forked from Karylab-cklius/vllm
Compare commits
1030
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6bf03e0d95 | ||
|
|
2595d5cebc | ||
|
|
ee5a89f4d7 | ||
|
|
e26264f3ef | ||
|
|
4c81772e8b | ||
|
|
27c3e579f0 | ||
|
|
8df14cfc8c | ||
|
|
370b678a02 | ||
|
|
5c0c987c03 | ||
|
|
5f8e73cb8b | ||
|
|
83762b77b0 | ||
|
|
a02984ed47 | ||
|
|
fc1c548093 | ||
|
|
481e481be7 | ||
|
|
8e981630c9 | ||
|
|
9a48eef89a | ||
|
|
1ef1c7ebba | ||
|
|
54503ecec0 | ||
|
|
0067311536 | ||
|
|
51878e5b6e | ||
|
|
76fedaa2a5 | ||
|
|
19069bcbd5 | ||
|
|
1bd8f80a64 | ||
|
|
0b6636cbcb | ||
|
|
4a6440acef | ||
|
|
bec0a4ede6 | ||
|
|
3d99b0499a | ||
|
|
04d553f390 | ||
|
|
9c18e90f6c | ||
|
|
092387963c | ||
|
|
1bf3997eae | ||
|
|
29fd688892 | ||
|
|
ed908cf0a0 | ||
|
|
26ff616bbf | ||
|
|
f378f79b7c | ||
|
|
735def4fcf | ||
|
|
c227aaa3f8 | ||
|
|
08dfd68610 | ||
|
|
978a6dfa3f | ||
|
|
85c09e9885 | ||
|
|
b12cca6a23 | ||
|
|
e257faf87d | ||
|
|
fabec87f63 | ||
|
|
7614b88ebd | ||
|
|
68ea76e780 | ||
|
|
c241c7a2b0 | ||
|
|
e23b19309b | ||
|
|
f36284a8d2 | ||
|
|
424df4f65d | ||
|
|
074bdd0d99 | ||
|
|
216ee58780 | ||
|
|
433f291195 | ||
|
|
28eaf05d56 | ||
|
|
300e33797f | ||
|
|
5715fde12c | ||
|
|
e5588e49bc | ||
|
|
95ed0feaa5 | ||
|
|
2d814a0082 | ||
|
|
88e5e2c57b | ||
|
|
feb384ada2 | ||
|
|
a0f6d767e4 | ||
|
|
f1a5adddb8 | ||
|
|
cac3e70cd4 | ||
|
|
e12b91b032 | ||
|
|
766469a4c4 | ||
|
|
ea0fa34f49 | ||
|
|
bbb0f945ff | ||
|
|
2ded1b24e7 | ||
|
|
b0dec2a11b | ||
|
|
ff8d3488f2 | ||
|
|
2285cfca46 | ||
|
|
e08a915146 | ||
|
|
67e7ea8977 | ||
|
|
429f405748 | ||
|
|
753c5039f0 | ||
|
|
299d2b5655 | ||
|
|
85b3a7264b | ||
|
|
b83be00cdd | ||
|
|
412414d8e0 | ||
|
|
ae6170f874 | ||
|
|
e87521626f | ||
|
|
1cd75b3dd4 | ||
|
|
0206f10871 | ||
|
|
ab7961a14a | ||
|
|
a07765c6bd | ||
|
|
1171467e91 | ||
|
|
529af88842 | ||
|
|
b8c7c86533 | ||
|
|
2c17d33f42 | ||
|
|
bc44f9feb7 | ||
|
|
7802c20c4e | ||
|
|
95d6d6f4bb | ||
|
|
56da398dac | ||
|
|
26831949b4 | ||
|
|
6cf7b26bd4 | ||
|
|
5f85975624 | ||
|
|
dcdd756d75 | ||
|
|
0d2f4e7c9c | ||
|
|
49abadaedb | ||
|
|
089e412878 | ||
|
|
a5d19cbb95 | ||
|
|
8347c6e6e1 | ||
|
|
b2cf70ea3a | ||
|
|
d1f1d86797 | ||
|
|
f05603fa28 | ||
|
|
c2ecd0f888 | ||
|
|
0d12618e98 | ||
|
|
68b4a1d582 | ||
|
|
572b25b03e | ||
|
|
9f2b3b093c | ||
|
|
cd0de48d08 | ||
|
|
934eeaecfb | ||
|
|
2cae98dfa5 | ||
|
|
db39d60010 | ||
|
|
a1ab51afb6 | ||
|
|
e7b3853bac | ||
|
|
eeaf23107f | ||
|
|
285c08c036 | ||
|
|
1f4ad059d1 | ||
|
|
04a703e397 | ||
|
|
bd3bb4eb26 | ||
|
|
440002552e | ||
|
|
99a85617bf | ||
|
|
7c67da967f | ||
|
|
d79855eaac | ||
|
|
51e5372f3d | ||
|
|
7cc2e8e74f | ||
|
|
2c64b4c1cc | ||
|
|
d35eba302f | ||
|
|
c0e8e1f12a | ||
|
|
5d5fab0061 | ||
|
|
2afa3f7e95 | ||
|
|
80eb01e93d | ||
|
|
d9e57ea82e | ||
|
|
9021589498 | ||
|
|
0303f37a54 | ||
|
|
dd127d82ed | ||
|
|
0ca6eee743 | ||
|
|
5e975eae1a | ||
|
|
f7fc0ca993 | ||
|
|
f7efab58ec | ||
|
|
e97c3cb303 | ||
|
|
4aceabf8c1 | ||
|
|
6e35c5e5af | ||
|
|
aad0fb741b | ||
|
|
7d2ce5750e | ||
|
|
675f4295cd | ||
|
|
d99adcebdc | ||
|
|
c8c2f838e7 | ||
|
|
dd0d74cd92 | ||
|
|
55da232db6 | ||
|
|
3f99883d97 | ||
|
|
47c40bfe8a | ||
|
|
3dd910da42 | ||
|
|
7bd154375d | ||
|
|
2f3f441f84 | ||
|
|
d6875196ad | ||
|
|
abe41f28de | ||
|
|
c3284c31f5 | ||
|
|
c74e751824 | ||
|
|
b93cbd7416 | ||
|
|
bdc6f3bfa1 | ||
|
|
392d1b4d2e | ||
|
|
21b396abe1 | ||
|
|
bdaf27519f | ||
|
|
beb4327c46 | ||
|
|
c46ced1ee3 | ||
|
|
65dcde1695 | ||
|
|
65a7b46284 | ||
|
|
93e2ab7111 | ||
|
|
8b745527cd | ||
|
|
920469974a | ||
|
|
8b91cd5b20 | ||
|
|
dd94484577 | ||
|
|
7ff656cc8b | ||
|
|
0a2965b1b3 | ||
|
|
0ed05b6f82 | ||
|
|
ed051fab54 | ||
|
|
48fcfc926c | ||
|
|
3354dba381 | ||
|
|
cbb5f045be | ||
|
|
b3e85be663 | ||
|
|
d3e69fd671 | ||
|
|
c85d72076a | ||
|
|
c5b66233b2 | ||
|
|
066f02ae94 | ||
|
|
5d23ca47ab | ||
|
|
e55cc59e52 | ||
|
|
ba50b9763f | ||
|
|
b4cfbc24d3 | ||
|
|
1e823dc01d | ||
|
|
8e61b646e2 | ||
|
|
e040899a00 | ||
|
|
dd5c299fbe | ||
|
|
6db31c8e76 | ||
|
|
cbe9c40f99 | ||
|
|
2f71b2bd9f | ||
|
|
32ab064621 | ||
|
|
c64c356990 | ||
|
|
34e6dfced8 | ||
|
|
39a1d32b59 | ||
|
|
700e882eab | ||
|
|
a4f019fa25 | ||
|
|
9dd2465896 | ||
|
|
a46c9329e5 | ||
|
|
445321fab4 | ||
|
|
69f3150981 | ||
|
|
86db6c3070 | ||
|
|
5769a7382c | ||
|
|
8484ca5d45 | ||
|
|
482e5524fe | ||
|
|
567a78432d | ||
|
|
d891b9bd51 | ||
|
|
04adc8843b | ||
|
|
ae098abe3f | ||
|
|
b1384f5ec6 | ||
|
|
b136cc2c2c | ||
|
|
9fde043f54 | ||
|
|
24dd2aec81 | ||
|
|
3ee9eea928 | ||
|
|
5bce653e09 | ||
|
|
5ad11172b7 | ||
|
|
f70caef48b | ||
|
|
8d8ec38361 | ||
|
|
b1c6dba558 | ||
|
|
598d51153a | ||
|
|
095adf1fdc | ||
|
|
51ee564e56 | ||
|
|
373eb314af | ||
|
|
641cb59592 | ||
|
|
07f9baf756 | ||
|
|
7a90eb98ab | ||
|
|
8f4c69b222 | ||
|
|
8b79971bb9 | ||
|
|
f676808ba0 | ||
|
|
98e4726a14 | ||
|
|
740f379fae | ||
|
|
40cc2e8327 | ||
|
|
ba22152096 | ||
|
|
90ce3a09be | ||
|
|
26c754d847 | ||
|
|
3d7f357ebf | ||
|
|
736f1a5907 | ||
|
|
344609ab17 | ||
|
|
d039c17114 | ||
|
|
cdab28319f | ||
|
|
2fa10566e3 | ||
|
|
fb265fc8fb | ||
|
|
8f0e75e16b | ||
|
|
98ba9b9583 | ||
|
|
990c2a0187 | ||
|
|
e433634c78 | ||
|
|
16f8110935 | ||
|
|
d9c1767cd4 | ||
|
|
e9cc1fd093 | ||
|
|
f1073c050c | ||
|
|
394edc8108 | ||
|
|
69715823df | ||
|
|
6569df6a3e | ||
|
|
f2aaf59151 | ||
|
|
95a248faed | ||
|
|
d2ec433e37 | ||
|
|
78a04c208d | ||
|
|
b71218107f | ||
|
|
cc1d020d01 | ||
|
|
8974ed89cd | ||
|
|
fb2faceacd | ||
|
|
b6cc46ec3b | ||
|
|
fa4321de3d | ||
|
|
9226613043 | ||
|
|
34b560b725 | ||
|
|
91b5647300 | ||
|
|
4a6bf3c77f | ||
|
|
d2afe39647 | ||
|
|
2a9113f998 | ||
|
|
0cd6f767e3 | ||
|
|
f1445f6dbd | ||
|
|
1d354c694e | ||
|
|
2f21224527 | ||
|
|
fa1fa968c4 | ||
|
|
6eac8e0070 | ||
|
|
1a308c449c | ||
|
|
e7c9df9449 | ||
|
|
26eb87204d | ||
|
|
4c3c17d43b | ||
|
|
f329ce405b | ||
|
|
07516fda67 | ||
|
|
67ff0ae30f | ||
|
|
ab3b6d97aa | ||
|
|
fb5291b35b | ||
|
|
d6d39c111e | ||
|
|
379950191f | ||
|
|
576bf75d0e | ||
|
|
f006e5a24c | ||
|
|
f63dca6838 | ||
|
|
8651f043b8 | ||
|
|
3775d5fcab | ||
|
|
d7192cfccf | ||
|
|
978de83353 | ||
|
|
a14f57a3ac | ||
|
|
18f658bb31 | ||
|
|
400a9c386d | ||
|
|
bbdcbe4686 | ||
|
|
4875b4456b | ||
|
|
1f486d96a1 | ||
|
|
b790c84cde | ||
|
|
6429d5f527 | ||
|
|
fbc9ba6d30 | ||
|
|
2dfaae752b | ||
|
|
bd8d9021ce | ||
|
|
3f0b773b30 | ||
|
|
9b8e76589d | ||
|
|
1aeabec355 | ||
|
|
979f5511d7 | ||
|
|
41de1380c2 | ||
|
|
d85601c20f | ||
|
|
276b837dc4 | ||
|
|
34bf7b45a0 | ||
|
|
4c3c64fcf7 | ||
|
|
442ccc6098 | ||
|
|
6768fbc76f | ||
|
|
407f406300 | ||
|
|
e24d1b24fe | ||
|
|
d29125c085 | ||
|
|
d715b3aa1e | ||
|
|
258f8de91f | ||
|
|
e392bf7a68 | ||
|
|
443e68cfa6 | ||
|
|
320ee285c9 | ||
|
|
ec0ffaacc8 | ||
|
|
178fd56094 | ||
|
|
a47f38f825 | ||
|
|
3e158ae62d | ||
|
|
a2f713002d | ||
|
|
de2a8fc042 | ||
|
|
84b9c2762f | ||
|
|
25fcb65d51 | ||
|
|
08a8a4af3f | ||
|
|
b0b8a286dd | ||
|
|
3af8789559 | ||
|
|
8357226f4f | ||
|
|
2665ed704b | ||
|
|
09663abde0 | ||
|
|
d63c8e9444 | ||
|
|
1360c42fe6 | ||
|
|
d0a2584773 | ||
|
|
7fe7fa9cda | ||
|
|
2b753ad200 | ||
|
|
e196268bad | ||
|
|
e91f5f8439 | ||
|
|
fa248139a0 | ||
|
|
d3229431f9 | ||
|
|
4787f2dd1b | ||
|
|
8cfeb84dba | ||
|
|
5fd442187c | ||
|
|
00eb7cefa3 | ||
|
|
c8bdcc0116 | ||
|
|
f5a8d73377 | ||
|
|
63fcce4de1 | ||
|
|
c638f9216a | ||
|
|
13c49f9845 | ||
|
|
f1cf6b0086 | ||
|
|
a78c15616f | ||
|
|
5c4db60f01 | ||
|
|
4e5ca89cfe | ||
|
|
a22e0dfc69 | ||
|
|
cc56379e28 | ||
|
|
024b06b0dc | ||
|
|
e7d0fcbc09 | ||
|
|
aa8bb5562e | ||
|
|
fa4bec9056 | ||
|
|
dee5da1dec | ||
|
|
ed41aa270a | ||
|
|
77a9c5ae28 | ||
|
|
f651a8a9a4 | ||
|
|
8f82be5705 | ||
|
|
a461070d1c | ||
|
|
4470ae84de | ||
|
|
697c34b97b | ||
|
|
5b431b905c | ||
|
|
89e99202f2 | ||
|
|
b446792306 | ||
|
|
c3b1f9e827 | ||
|
|
df802a87b7 | ||
|
|
93d8f834dd | ||
|
|
aeb35b90f0 | ||
|
|
9a08a5118e | ||
|
|
c5200d3565 | ||
|
|
3c1396bab6 | ||
|
|
9969466a59 | ||
|
|
3406e8f83d | ||
|
|
a264e41975 | ||
|
|
f098ee70c7 | ||
|
|
9294dd27eb | ||
|
|
b1190d03cc | ||
|
|
92c7fac640 | ||
|
|
ac521f6237 | ||
|
|
28242824e0 | ||
|
|
68294739d1 | ||
|
|
c8d2f3cb14 | ||
|
|
345b28ff2f | ||
|
|
248d1fbb71 | ||
|
|
11b26c5528 | ||
|
|
20434c472e | ||
|
|
c8f9c156a5 | ||
|
|
953bba488d | ||
|
|
3a9784b82c | ||
|
|
3cecee40f3 | ||
|
|
a7732537f4 | ||
|
|
727971f1c1 | ||
|
|
25671cb520 | ||
|
|
27d5f78b63 | ||
|
|
7a341fa109 | ||
|
|
f41e8ddc97 | ||
|
|
245888ff77 | ||
|
|
e840f0d3f5 | ||
|
|
fcaa84efa7 | ||
|
|
9e84ec8648 | ||
|
|
d8f483dc30 | ||
|
|
dc148dc4d7 | ||
|
|
7cf7cbcd95 | ||
|
|
c231d1f290 | ||
|
|
db808b3961 | ||
|
|
00ebf19cca | ||
|
|
ded6676458 | ||
|
|
7a327f0b4f | ||
|
|
1ab9522935 | ||
|
|
0fc2512094 | ||
|
|
62c7d8009f | ||
|
|
ab80b3dff4 | ||
|
|
91055efd36 | ||
|
|
3675bcff67 | ||
|
|
bdbd7278b6 | ||
|
|
5dc36a4fa5 | ||
|
|
aab7af0bcb | ||
|
|
536047755e | ||
|
|
1907d3854a | ||
|
|
ea9ddf59fc | ||
|
|
8cf7c4d8ad | ||
|
|
8e9d70fdd5 | ||
|
|
364ee36af1 | ||
|
|
06fae69114 | ||
|
|
14f8660a18 | ||
|
|
aed541def4 | ||
|
|
2bc20e8aba | ||
|
|
8cc242335d | ||
|
|
ba22cb6765 | ||
|
|
81bcced482 | ||
|
|
fb42e5219e | ||
|
|
0feca7ffa8 | ||
|
|
97b5ce5c39 | ||
|
|
4236514098 | ||
|
|
e45c8a9f4b | ||
|
|
b153dd3f28 | ||
|
|
930f8dc0a1 | ||
|
|
a16dbd5b85 | ||
|
|
bec232a914 | ||
|
|
b5c9e1ac33 | ||
|
|
ae2c4f3db7 | ||
|
|
fca432e60a | ||
|
|
af1ee8c475 | ||
|
|
5b4cb69523 | ||
|
|
9fc0c08026 | ||
|
|
f2b5fabb23 | ||
|
|
b8cb75b149 | ||
|
|
43916891b2 | ||
|
|
cda05ee8c4 | ||
|
|
77654d080c | ||
|
|
75698e60b3 | ||
|
|
8632c884dc | ||
|
|
c3734e8334 | ||
|
|
53f7553f09 | ||
|
|
4eb227992a | ||
|
|
ebcf511ec3 | ||
|
|
8fc1b2d046 | ||
|
|
5316638a5e | ||
|
|
61ab70ec3b | ||
|
|
a309d4fe60 | ||
|
|
72f639927f | ||
|
|
8ad4a01825 | ||
|
|
7be582697b | ||
|
|
030c9523bd | ||
|
|
4708292d48 | ||
|
|
debec6440b | ||
|
|
c8fb2963bd | ||
|
|
379acd4e4f | ||
|
|
07d33e575b | ||
|
|
36bbecd643 | ||
|
|
6149187a4c | ||
|
|
49e28e8e91 | ||
|
|
0ca39c4f1f | ||
|
|
6185d73882 | ||
|
|
bc8481af09 | ||
|
|
59575da46d | ||
|
|
3483240b7e | ||
|
|
eddfd4cf21 | ||
|
|
a4e3cb40d0 | ||
|
|
ab132ee98b | ||
|
|
e186107870 | ||
|
|
0e207dac78 | ||
|
|
9e86352c60 | ||
|
|
5051698e41 | ||
|
|
db28ae2d07 | ||
|
|
f6bb8682ee | ||
|
|
4559c43a95 | ||
|
|
5274c1181d | ||
|
|
58d6a6e60a | ||
|
|
a2abce646f | ||
|
|
311ad689ad | ||
|
|
0472436541 | ||
|
|
4dfbf1503b | ||
|
|
95528527ea | ||
|
|
c2127a25c7 | ||
|
|
03c6d01c30 | ||
|
|
4b643c463e | ||
|
|
7544286b04 | ||
|
|
89876b0c54 | ||
|
|
5c91039c41 | ||
|
|
5ecae3266c | ||
|
|
6eb63a1da6 | ||
|
|
09841ae705 | ||
|
|
a2a92cbbaa | ||
|
|
35e6c86caa | ||
|
|
c7ca0bccae | ||
|
|
c6741b2ad4 | ||
|
|
a65f93fb2e | ||
|
|
11a12305c0 | ||
|
|
798185d438 | ||
|
|
9036c89ee4 | ||
|
|
b6caeb5a09 | ||
|
|
8bf064f8d3 | ||
|
|
ea2ead1db3 | ||
|
|
56aa067bf0 | ||
|
|
35e3850fa9 | ||
|
|
51a99565c3 | ||
|
|
867fd5e8ed | ||
|
|
9fd00ee006 | ||
|
|
091d13976c | ||
|
|
b588f66dc2 | ||
|
|
455f25aa13 | ||
|
|
d706dec904 | ||
|
|
68ee8300a0 | ||
|
|
ddd3855a28 | ||
|
|
00e045b7c7 | ||
|
|
17a71d8702 | ||
|
|
2e058851d3 | ||
|
|
1a92dfcce4 | ||
|
|
d0f800811b | ||
|
|
c6dd32a810 | ||
|
|
af16446bf3 | ||
|
|
3f67477497 | ||
|
|
1d41009e81 | ||
|
|
b94f212e37 | ||
|
|
d8eb734d94 | ||
|
|
2ff76a5e85 | ||
|
|
75fdcc82a5 | ||
|
|
77f8796d16 | ||
|
|
c40d307731 | ||
|
|
65e655d295 | ||
|
|
6e2fb02fe5 | ||
|
|
274325dd43 | ||
|
|
95e6442a6b | ||
|
|
701a23d99f | ||
|
|
dccb412e2c | ||
|
|
c6554f321c | ||
|
|
3d3b96488f | ||
|
|
658b54efe4 | ||
|
|
abc71548ef | ||
|
|
4e07ca2c92 | ||
|
|
e71bc6da85 | ||
|
|
37ce34922f | ||
|
|
c2507fb293 | ||
|
|
8921c4be88 | ||
|
|
8e394244a5 | ||
|
|
302954e5f6 | ||
|
|
950ee4c2e4 | ||
|
|
d980a3cc6e | ||
|
|
bf292b5f6b | ||
|
|
5e3dad04b1 | ||
|
|
63e161f296 | ||
|
|
c7645bce04 | ||
|
|
35a49fcfc2 | ||
|
|
915e99ec67 | ||
|
|
5b33041746 | ||
|
|
1a4984520e | ||
|
|
e312c5cb25 | ||
|
|
1502cf6274 | ||
|
|
d350fa8ddd | ||
|
|
dbc49b6b99 | ||
|
|
552a9dbe59 | ||
|
|
02a1f23711 | ||
|
|
652d962bc9 | ||
|
|
5314665bad | ||
|
|
3daea7ceb9 | ||
|
|
cc7981599e | ||
|
|
32bb3195f0 | ||
|
|
ad28d605e6 | ||
|
|
ae7c8ec223 | ||
|
|
1d3f4cb3a4 | ||
|
|
f9e684499f | ||
|
|
c53994e134 | ||
|
|
27da2a2ac4 | ||
|
|
a2e8ec3d52 | ||
|
|
e8c24a7695 | ||
|
|
2a6f8f0c05 | ||
|
|
c5e3c40877 | ||
|
|
8b4d93ba2b | ||
|
|
e8e7b592d1 | ||
|
|
e53a17232c | ||
|
|
96eb8ddc41 | ||
|
|
8fa36fbbeb | ||
|
|
e45b279928 | ||
|
|
d490b98162 | ||
|
|
1744adc256 | ||
|
|
cdfa2fd7e9 | ||
|
|
6f3da461d1 | ||
|
|
d3130d878c | ||
|
|
9bfd878a48 | ||
|
|
2365b7a8e7 | ||
|
|
15be78732b | ||
|
|
92221485aa | ||
|
|
a6f41ab678 | ||
|
|
c63cd4906c | ||
|
|
638b1a99cc | ||
|
|
72adb20a6a | ||
|
|
2396d91e93 | ||
|
|
9b215ae60b | ||
|
|
4d3b4b9b01 | ||
|
|
77c1d9fe9b | ||
|
|
36fd7e8b86 | ||
|
|
fc61c6fc26 | ||
|
|
e2af449c39 | ||
|
|
3f5a1e1733 | ||
|
|
710ebaa189 | ||
|
|
1aad125815 | ||
|
|
dc55936f64 | ||
|
|
76c3c4ff63 | ||
|
|
efb5acffd5 | ||
|
|
6e3a983cf3 | ||
|
|
1273a8f05a | ||
|
|
9e88e969c0 | ||
|
|
dda3aca47f | ||
|
|
23aed9b0ee | ||
|
|
cd347298e8 | ||
|
|
b69816043a | ||
|
|
fc7fc421e9 | ||
|
|
e06a83445c | ||
|
|
d7ab9be775 | ||
|
|
6a1570711c | ||
|
|
d6696e2385 | ||
|
|
84c2f9f0fb | ||
|
|
49f2104c53 | ||
|
|
d511b5bae9 | ||
|
|
3c43237233 | ||
|
|
56ca5997ea | ||
|
|
cf57311187 | ||
|
|
e7df232288 | ||
|
|
b3a688cb9e | ||
|
|
1cd3e0e945 | ||
|
|
f889325c51 | ||
|
|
bb61177e49 | ||
|
|
7f99e80c3b | ||
|
|
2801b11156 | ||
|
|
007b5a52ed | ||
|
|
24d5186138 | ||
|
|
7dc036058b | ||
|
|
61ee183d28 | ||
|
|
84c62e1cbd | ||
|
|
061043eaca | ||
|
|
93ec645878 | ||
|
|
563c628968 | ||
|
|
0bc479e6eb | ||
|
|
62890e204c | ||
|
|
a2cb08b3d5 | ||
|
|
cf9fd6457e | ||
|
|
d4448b511d | ||
|
|
f1a6703edd | ||
|
|
160c80a34c | ||
|
|
f237e16b41 | ||
|
|
70749fdcca | ||
|
|
d20dbf921b | ||
|
|
ede54b926e | ||
|
|
52fbe12283 | ||
|
|
dc0d318177 | ||
|
|
d7c1821b5a | ||
|
|
4cd1a84c88 | ||
|
|
191826ec61 | ||
|
|
549c7074cd | ||
|
|
489abadfb8 | ||
|
|
96de8bb389 | ||
|
|
9d6fdc2901 | ||
|
|
4c5bc41ba6 | ||
|
|
ac1fa74616 | ||
|
|
556bc4e3a0 | ||
|
|
05a0caba91 | ||
|
|
7ee4d22009 | ||
|
|
ce9f64020b | ||
|
|
4ed8eaafb0 | ||
|
|
6af0559ddb | ||
|
|
e2bdc24612 | ||
|
|
bcbeaac786 | ||
|
|
e48f2aa4ca | ||
|
|
d86c66c981 | ||
|
|
80e511772f | ||
|
|
855cd4d787 | ||
|
|
3cc871aaf1 | ||
|
|
0a3e2dbc09 | ||
|
|
84f13374b3 | ||
|
|
b28103e1ca | ||
|
|
abc33134fa | ||
|
|
6617db1bfb | ||
|
|
899d72a58c | ||
|
|
11b56b2ff2 | ||
|
|
0d4d164488 | ||
|
|
0775b882ba | ||
|
|
7c2e08451a | ||
|
|
ef361de916 | ||
|
|
acce57d8dd | ||
|
|
68afd78897 | ||
|
|
37a682d392 | ||
|
|
d8e422ccda | ||
|
|
e368415daa | ||
|
|
ceae5bcbda | ||
|
|
6691f087a6 | ||
|
|
f4d5f73ffa | ||
|
|
fd50a66015 | ||
|
|
84586c9acc | ||
|
|
40e5522121 | ||
|
|
f3410b3bb1 | ||
|
|
568874fec2 | ||
|
|
275b43183c | ||
|
|
547d2c40d7 | ||
|
|
2aaaf3febd | ||
|
|
156b12667c | ||
|
|
9f6f296428 | ||
|
|
e51e700470 | ||
|
|
f59db63732 | ||
|
|
9f5117820f | ||
|
|
1bf149f334 | ||
|
|
2a675a7b9f | ||
|
|
7d47cff933 | ||
|
|
091bc1026e | ||
|
|
3554ada5d8 | ||
|
|
31ca9504b1 | ||
|
|
d32575a2d2 | ||
|
|
83fa302ca4 | ||
|
|
20b5af55c1 | ||
|
|
901a3b091c | ||
|
|
2d721ab5d8 | ||
|
|
accaa434f3 | ||
|
|
a04654da23 | ||
|
|
25bc3be49c | ||
|
|
a46f3eb232 | ||
|
|
6c427dd401 | ||
|
|
3ce5823762 | ||
|
|
04c2a8deac | ||
|
|
7e47fb72b5 | ||
|
|
a8481be7a9 | ||
|
|
9d3317172c | ||
|
|
430a95ae3a | ||
|
|
56e5797511 | ||
|
|
8db12169a4 | ||
|
|
33f50773cb | ||
|
|
fa36f86d77 | ||
|
|
8207ce0850 | ||
|
|
e48592066e | ||
|
|
91ba720b75 | ||
|
|
6ead164e52 | ||
|
|
c97e8f99d6 | ||
|
|
183b5f27ea | ||
|
|
ca5b24695b | ||
|
|
6f6bd3b8fe | ||
|
|
70ef4d3009 | ||
|
|
e2fe837572 | ||
|
|
fbf9ff7cf4 | ||
|
|
6cc2c9ba3a | ||
|
|
c0b2d8f471 | ||
|
|
d1a38c2762 | ||
|
|
2b4a7491ec | ||
|
|
82ede09a5a | ||
|
|
fbf520cf3a | ||
|
|
44d95069e9 | ||
|
|
3ce15fd574 | ||
|
|
e4b3da3feb | ||
|
|
3e6529cc0e | ||
|
|
ac614587f5 | ||
|
|
f2069b005b | ||
|
|
ccd49f6821 | ||
|
|
1c7bc18318 | ||
|
|
9a938df64e | ||
|
|
3da4a1b124 | ||
|
|
6871738777 | ||
|
|
aa4990a9a2 | ||
|
|
a4610da0c6 | ||
|
|
09cdcf34aa | ||
|
|
d2c671c29b | ||
|
|
b5a2adec4b | ||
|
|
78739e3bda | ||
|
|
89accad2cc | ||
|
|
3c8e49596c | ||
|
|
cec2ec1176 | ||
|
|
435f82d61a | ||
|
|
1c4b51b990 | ||
|
|
2e2c47928b | ||
|
|
80abe0de7d | ||
|
|
a9f7b2d41c | ||
|
|
d14e551a53 | ||
|
|
68567ef2df | ||
|
|
6bc6f2d86d | ||
|
|
1eb2cc961e | ||
|
|
31124749d1 | ||
|
|
9037498c22 | ||
|
|
db32b53e30 | ||
|
|
b529bfd6c5 | ||
|
|
f3df7a7231 | ||
|
|
485bbe1c6f | ||
|
|
a19ff2218a | ||
|
|
4f0d0049a0 | ||
|
|
13b83d77ad | ||
|
|
50241602fd | ||
|
|
12fe2a9aac | ||
|
|
89bd2c14d3 | ||
|
|
9c450b1027 | ||
|
|
635c38338a | ||
|
|
c441ad1c07 | ||
|
|
745bba5ea8 | ||
|
|
2cac89f9da | ||
|
|
3e6e33526d | ||
|
|
b91b7726e0 | ||
|
|
d3ad8e8bcd | ||
|
|
b80ce9dd2f | ||
|
|
b5495cc5f9 | ||
|
|
183a430c13 | ||
|
|
a346d589f5 | ||
|
|
7df3d7dada | ||
|
|
8dd1b702f2 | ||
|
|
f57ac274b2 | ||
|
|
6e919960af | ||
|
|
c88d3d4775 | ||
|
|
ab7fcbdd5d | ||
|
|
3b4a76b63f | ||
|
|
cc22621b51 | ||
|
|
77148992cf | ||
|
|
891cc4b9c5 | ||
|
|
1bdf9810aa | ||
|
|
ebfbcfe46a | ||
|
|
e9de72fe6c | ||
|
|
d272418f45 | ||
|
|
7ff7f5c8eb | ||
|
|
dced290769 | ||
|
|
93bad11912 | ||
|
|
0fbf42af84 | ||
|
|
e6cd8913dd | ||
|
|
859e4d436b | ||
|
|
4a083cc858 | ||
|
|
ca7e1f2c43 | ||
|
|
dec860fb19 | ||
|
|
0a49fb2b13 | ||
|
|
4a8abf37c7 | ||
|
|
01192139bf | ||
|
|
b9a7cd464c | ||
|
|
69bdd34542 | ||
|
|
ec67d7ae61 | ||
|
|
ecf9d83520 | ||
|
|
c9135db27c | ||
|
|
2a6c6b9429 | ||
|
|
ab66606993 | ||
|
|
9ea3a4015b | ||
|
|
560fb8b867 | ||
|
|
675cd5d228 | ||
|
|
7f616c327d | ||
|
|
c3c6d723fd | ||
|
|
41dcf49ca5 | ||
|
|
35e4dd4a69 | ||
|
|
4ce2d01453 | ||
|
|
16908e132e | ||
|
|
225936a1dd | ||
|
|
f6ba720963 | ||
|
|
b53b1c7ffe | ||
|
|
79ca54d221 | ||
|
|
09f3cd5c10 | ||
|
|
ea6078fe6a | ||
|
|
a0df04e477 | ||
|
|
e2352c2974 | ||
|
|
25faa1f4cc | ||
|
|
4583630b56 | ||
|
|
21da47dabe | ||
|
|
6c379b9e54 | ||
|
|
5099474633 | ||
|
|
058cc0a8b6 | ||
|
|
837db7605e | ||
|
|
bf2a393034 | ||
|
|
d682968aa9 | ||
|
|
021cdf72bc | ||
|
|
4cb5e746b6 | ||
|
|
22cc891108 | ||
|
|
afdcbd5d39 | ||
|
|
8d4f54966c | ||
|
|
351c72d6e5 | ||
|
|
7299e6509e | ||
|
|
08985351f3 | ||
|
|
5fd3b276f8 | ||
|
|
1e9f04da14 | ||
|
|
702214146c | ||
|
|
a331589394 | ||
|
|
e945169207 | ||
|
|
554352a311 | ||
|
|
421c1ec448 | ||
|
|
b4c80ec0fd | ||
|
|
f428718ffe | ||
|
|
4403af8fb5 | ||
|
|
d57888efa4 | ||
|
|
ed938ad7db | ||
|
|
731fb3323d | ||
|
|
8dd8b6ed78 | ||
|
|
e1a5fc406b | ||
|
|
b4092176b9 | ||
|
|
ebbb2d55ac | ||
|
|
2959a9273a | ||
|
|
1797576237 | ||
|
|
0d339cf135 | ||
|
|
5fd21eb0b2 | ||
|
|
9d4b87f4f0 | ||
|
|
58b2e89642 | ||
|
|
2659f60a1a | ||
|
|
091386a99b | ||
|
|
d112eb1ac7 | ||
|
|
2a47a9ff0f | ||
|
|
9c7c74bf10 | ||
|
|
5e27b2baf4 | ||
|
|
eb0fdeb1e8 | ||
|
|
46f74e144b | ||
|
|
0a7bacdcac | ||
|
|
8b2b566ea7 | ||
|
|
0b131b16c9 | ||
|
|
bcb518ad7a | ||
|
|
06e1e0885c | ||
|
|
1a59078c87 | ||
|
|
fa85ead2f3 | ||
|
|
e28e8c8782 | ||
|
|
ee0fd6984a | ||
|
|
d537122398 | ||
|
|
3d20275bb4 | ||
|
|
f694d43b33 | ||
|
|
3c6084bb0d | ||
|
|
68ff30d40e | ||
|
|
6d8fff5698 | ||
|
|
e2c58570ea | ||
|
|
43fa24e832 | ||
|
|
93bbe94d3a | ||
|
|
17bc144556 | ||
|
|
295232a26a | ||
|
|
56e4345226 | ||
|
|
e9993a52aa | ||
|
|
a46abb7ae6 | ||
|
|
4c62663315 | ||
|
|
d78650cf97 | ||
|
|
5bdc01bcc3 | ||
|
|
20a5f8b43b | ||
|
|
7b5d60cc37 | ||
|
|
14b438a98b | ||
|
|
2785a5e0e6 | ||
|
|
efd15e192a | ||
|
|
556b063e45 | ||
|
|
aa0ac8a661 | ||
|
|
b831374cf1 | ||
|
|
71bc19dbdd | ||
|
|
ef2c40dc00 | ||
|
|
4bf699d310 | ||
|
|
520828789c | ||
|
|
9d4dc4ca2f | ||
|
|
b9684d99e9 | ||
|
|
4fadf9c92c | ||
|
|
d8d95998dc | ||
|
|
475a6ad18a | ||
|
|
f2beaa80c8 | ||
|
|
8e27a9c215 | ||
|
|
7d567172fc | ||
|
|
f00e163f35 | ||
|
|
44b2512767 | ||
|
|
188c68798e | ||
|
|
c45f681932 | ||
|
|
89e8645a9e | ||
|
|
88a9cdd439 | ||
|
|
6f612fbedf | ||
|
|
506ec6d656 | ||
|
|
a52205bccf | ||
|
|
3d34f8cbdc | ||
|
|
eb04c769d3 | ||
|
|
ce3ef17bec | ||
|
|
bf5149b516 | ||
|
|
cca3365b73 | ||
|
|
040df8f2ea | ||
|
|
ced32bb474 | ||
|
|
c5e5c33fcd | ||
|
|
a8c86eeb16 | ||
|
|
7e179e4bc0 | ||
|
|
405c7cf283 | ||
|
|
3f53e2138f | ||
|
|
d53f4593ce | ||
|
|
ad32608e24 | ||
|
|
b2cfae777d | ||
|
|
3f1ff1ff14 | ||
|
|
c69c73418a | ||
|
|
ebf3a6d705 | ||
|
|
c4fd9794e9 | ||
|
|
7ad894c86a | ||
|
|
a7fdfeef72 | ||
|
|
8bf374955f | ||
|
|
9096659edb | ||
|
|
81d8f4ebac | ||
|
|
a9a8a32dcd | ||
|
|
9d808e2309 | ||
|
|
f3858d5422 | ||
|
|
259ff891be | ||
|
|
6607a80dab | ||
|
|
b8bd773fe4 | ||
|
|
2addbb9cc9 | ||
|
|
e3cfea2e1b | ||
|
|
f99260d2aa | ||
|
|
3f65e21e32 | ||
|
|
b00e76ff72 | ||
|
|
f4359a70f9 | ||
|
|
3afe659b6b | ||
|
|
16e91176cf | ||
|
|
ab8b0fe338 | ||
|
|
d467a2a7f2 | ||
|
|
76a373eff4 | ||
|
|
25ee659db0 | ||
|
|
eacff17c8d |
@@ -2,17 +2,17 @@ name: vllm_intel_ci
|
|||||||
job_dirs:
|
job_dirs:
|
||||||
- ".buildkite/intel_jobs"
|
- ".buildkite/intel_jobs"
|
||||||
run_all_patterns:
|
run_all_patterns:
|
||||||
|
- ".buildkite/ci_config_intel.yaml"
|
||||||
|
- ".buildkite/scripts/hardware_ci/run-intel-test.sh"
|
||||||
- "docker/Dockerfile"
|
- "docker/Dockerfile"
|
||||||
|
- "docker/Dockerfile.xpu"
|
||||||
- "CMakeLists.txt"
|
- "CMakeLists.txt"
|
||||||
- "requirements/common.txt"
|
- "requirements/common.txt"
|
||||||
- "requirements/xpu.txt"
|
- "requirements/xpu.txt"
|
||||||
- "requirements/build/cuda.txt"
|
|
||||||
- "requirements/test/cuda.txt"
|
|
||||||
- "setup.py"
|
- "setup.py"
|
||||||
- "csrc/"
|
- "csrc/"
|
||||||
- "cmake/"
|
- "cmake/"
|
||||||
run_all_exclude_patterns:
|
run_all_exclude_patterns:
|
||||||
- "docker/Dockerfile."
|
|
||||||
- "csrc/cpu/"
|
- "csrc/cpu/"
|
||||||
- "csrc/rocm/"
|
- "csrc/rocm/"
|
||||||
- "cmake/hipify.py"
|
- "cmake/hipify.py"
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ run_all_patterns:
|
|||||||
- "docker/docker-bake-rocm.hcl"
|
- "docker/docker-bake-rocm.hcl"
|
||||||
- ".buildkite/hardware_tests/amd.yaml"
|
- ".buildkite/hardware_tests/amd.yaml"
|
||||||
- ".buildkite/scripts/ci-bake-rocm.sh"
|
- ".buildkite/scripts/ci-bake-rocm.sh"
|
||||||
|
- ".buildkite/scripts/rocm/"
|
||||||
- ".buildkite/scripts/hardware_ci/run-amd-test.py"
|
- ".buildkite/scripts/hardware_ci/run-amd-test.py"
|
||||||
- ".buildkite/scripts/hardware_ci/run-amd-test.sh"
|
- ".buildkite/scripts/hardware_ci/run-amd-test.sh"
|
||||||
- "CMakeLists.txt"
|
- "CMakeLists.txt"
|
||||||
|
|||||||
@@ -1,18 +1,45 @@
|
|||||||
group: Hardware - AMD Build
|
group: Hardware - AMD Build
|
||||||
|
|
||||||
|
# ROCm image flow:
|
||||||
|
# 1. Refresh the long-lived ROCm base image only when Dockerfile.rocm_base changes.
|
||||||
|
# 2. Build ci_base from either the stable base or the freshly refreshed base.
|
||||||
|
# 3. Build the per-commit ROCm CI image and smoke-test it before GPU jobs run.
|
||||||
steps:
|
steps:
|
||||||
|
- label: "AMD: :docker: refresh ROCm base"
|
||||||
|
key: refresh-rocm-base-amd
|
||||||
|
depends_on: []
|
||||||
|
device: amd_cpu
|
||||||
|
no_plugin: true
|
||||||
|
commands:
|
||||||
|
- bash .buildkite/scripts/rocm/refresh-base-image.sh
|
||||||
|
env:
|
||||||
|
DOCKER_BUILDKIT: "1"
|
||||||
|
BUILDKIT_PROGRESS: "tty"
|
||||||
|
TERM: "xterm-256color"
|
||||||
|
retry:
|
||||||
|
automatic:
|
||||||
|
- exit_status: -1 # Agent was lost
|
||||||
|
limit: 1
|
||||||
|
- exit_status: -10 # Agent was lost
|
||||||
|
limit: 1
|
||||||
|
|
||||||
# Ensure ci_base is up-to-date before building the test image.
|
# Ensure ci_base is up-to-date before building the test image.
|
||||||
# Compares a content hash of ci_base-affecting files against the remote
|
# Compares a content hash of ci_base-affecting files against the remote
|
||||||
# image label. If hashes match the build is skipped (< 30 s); if they
|
# image label. If hashes match the build is skipped (< 30 s); if they
|
||||||
# differ ci_base is rebuilt and pushed automatically.
|
# differ ci_base is rebuilt and pushed automatically.
|
||||||
- label: "AMD: :docker: ensure ci_base"
|
- label: "AMD: :docker: ensure ci_base"
|
||||||
key: ensure-ci-base-amd
|
key: ensure-ci-base-amd
|
||||||
depends_on: []
|
soft_fail: false
|
||||||
|
depends_on:
|
||||||
|
- refresh-rocm-base-amd
|
||||||
device: amd_cpu
|
device: amd_cpu
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
commands:
|
commands:
|
||||||
- bash .buildkite/scripts/ci-bake-rocm.sh ci-base-rocm-ci-with-deps
|
- bash .buildkite/scripts/rocm/build-ci-base.sh
|
||||||
env:
|
env:
|
||||||
DOCKER_BUILDKIT: "1"
|
DOCKER_BUILDKIT: "1"
|
||||||
|
BUILDKIT_PROGRESS: "tty"
|
||||||
|
TERM: "xterm-256color"
|
||||||
VLLM_BAKE_FILE: "docker/docker-bake-rocm.hcl"
|
VLLM_BAKE_FILE: "docker/docker-bake-rocm.hcl"
|
||||||
PYTORCH_ROCM_ARCH: "gfx90a;gfx942;gfx950"
|
PYTORCH_ROCM_ARCH: "gfx90a;gfx942;gfx950"
|
||||||
REMOTE_VLLM: "1"
|
REMOTE_VLLM: "1"
|
||||||
@@ -26,40 +53,18 @@ steps:
|
|||||||
|
|
||||||
- label: "AMD: :docker: build test image and artifacts"
|
- label: "AMD: :docker: build test image and artifacts"
|
||||||
key: image-build-amd
|
key: image-build-amd
|
||||||
|
soft_fail: false
|
||||||
depends_on:
|
depends_on:
|
||||||
- ensure-ci-base-amd
|
- ensure-ci-base-amd
|
||||||
device: amd_cpu
|
device: amd_cpu
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
commands:
|
commands:
|
||||||
- |
|
- bash .buildkite/scripts/rocm/build-test-image.sh
|
||||||
if [[ "${ROCM_CI_ARTIFACT_ONLY:-0}" == "1" ]]; then
|
- bash .buildkite/scripts/rocm/smoke-test-image.sh
|
||||||
echo "ROCM_CI_ARTIFACT_ONLY=1; building ROCm wheel artifact only"
|
|
||||||
IMAGE_TAG="" bash .buildkite/scripts/ci-bake-rocm.sh test-rocm-ci-with-artifacts
|
|
||||||
else
|
|
||||||
bash .buildkite/scripts/ci-bake-rocm.sh test-rocm-ci-with-wheel
|
|
||||||
fi
|
|
||||||
- |
|
|
||||||
docker run --rm --network=none --entrypoint /bin/bash "rocm/vllm-ci:${BUILDKITE_COMMIT}" -ec '
|
|
||||||
if [ ! -d /vllm-workspace ]; then echo Missing directory: /vllm-workspace >&2; exit 1; fi
|
|
||||||
if [ ! -d /vllm-workspace/tests ]; then echo Missing directory: /vllm-workspace/tests >&2; exit 1; fi
|
|
||||||
if [ ! -d /vllm-workspace/src/vllm ]; then echo Missing directory: /vllm-workspace/src/vllm >&2; exit 1; fi
|
|
||||||
if [ ! -x /vllm-workspace/src/vllm/vllm-rs ]; then echo Missing executable: /vllm-workspace/src/vllm/vllm-rs >&2; exit 1; fi
|
|
||||||
command -v python3
|
|
||||||
command -v uv
|
|
||||||
command -v pytest
|
|
||||||
if ! command -v amd-smi >/dev/null 2>&1 && ! command -v rocminfo >/dev/null 2>&1; then
|
|
||||||
echo No ROCm CLI found in image >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
python3 - <<PY
|
|
||||||
import torch, vllm
|
|
||||||
print(torch.__version__)
|
|
||||||
print(vllm.__version__)
|
|
||||||
PY
|
|
||||||
echo AMD image smoke OK
|
|
||||||
'
|
|
||||||
env:
|
env:
|
||||||
DOCKER_BUILDKIT: "1"
|
DOCKER_BUILDKIT: "1"
|
||||||
|
BUILDKIT_PROGRESS: "tty"
|
||||||
|
TERM: "xterm-256color"
|
||||||
VLLM_BAKE_FILE: "docker/docker-bake-rocm.hcl"
|
VLLM_BAKE_FILE: "docker/docker-bake-rocm.hcl"
|
||||||
PYTORCH_ROCM_ARCH: "gfx90a;gfx942;gfx950"
|
PYTORCH_ROCM_ARCH: "gfx90a;gfx942;gfx950"
|
||||||
IMAGE_TAG: "rocm/vllm-ci:$BUILDKITE_COMMIT"
|
IMAGE_TAG: "rocm/vllm-ci:$BUILDKITE_COMMIT"
|
||||||
|
|||||||
@@ -17,12 +17,14 @@ steps:
|
|||||||
- tests/kernels/test_awq_int4_to_int8.py
|
- tests/kernels/test_awq_int4_to_int8.py
|
||||||
- tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
- tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
||||||
- tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
- tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
||||||
|
- tests/kernels/mamba/test_cpu_short_conv.py
|
||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 30m "
|
||||||
pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
|
pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
|
||||||
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
|
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
|
||||||
pytest -x -v -s tests/kernels/moe/test_cpu_quant_fused_moe.py
|
pytest -x -v -s tests/kernels/moe/test_cpu_quant_fused_moe.py
|
||||||
|
pytest -x -v -s tests/kernels/mamba/test_cpu_short_conv.py
|
||||||
pytest -x -v -s tests/kernels/test_onednn.py
|
pytest -x -v -s tests/kernels/test_onednn.py
|
||||||
pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py
|
pytest -x -v -s tests/kernels/test_awq_int4_to_int8.py
|
||||||
pytest -x -v -s tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
pytest -x -v -s tests/kernels/quantization/test_cpu_fp8_scaled_mm.py
|
||||||
@@ -53,7 +55,7 @@ steps:
|
|||||||
- tests/models/language/pooling/
|
- tests/models/language/pooling/
|
||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 40m "
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 50m "
|
||||||
pytest -x -v -s tests/models/language/generation -m cpu_model
|
pytest -x -v -s tests/models/language/generation -m cpu_model
|
||||||
pytest -x -v -s tests/models/language/pooling -m cpu_model"
|
pytest -x -v -s tests/models/language/pooling -m cpu_model"
|
||||||
|
|
||||||
@@ -68,13 +70,15 @@ steps:
|
|||||||
- vllm/v1/sample/ops/topk_topp_triton.py
|
- vllm/v1/sample/ops/topk_topp_triton.py
|
||||||
- vllm/v1/sample/ops/topk_topp_sampler.py
|
- vllm/v1/sample/ops/topk_topp_sampler.py
|
||||||
- tests/v1/sample/test_topk_topp_sampler.py
|
- tests/v1/sample/test_topk_topp_sampler.py
|
||||||
|
- tests/v1/e2e/test_cpu_linear_attn_chunked_prefix.py
|
||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
|
||||||
uv pip install git+https://github.com/triton-lang/triton-cpu.git@270e696d
|
uv pip install git+https://github.com/triton-lang/triton-cpu.git@270e696d
|
||||||
VLLM_USE_V2_MODEL_RUNNER=1 pytest -x -v -s tests/models/language/generation/test_granite.py -m cpu_model
|
VLLM_USE_V2_MODEL_RUNNER=1 pytest -x -v -s tests/models/language/generation/test_granite.py -m cpu_model
|
||||||
# TODO: move to CPU-Kernel Tests once triton-cpu has a pre-built wheel
|
# TODO: move to CPU-Kernel Tests once triton-cpu has a pre-built wheel
|
||||||
pytest -x -v -s tests/v1/sample/test_topk_topp_sampler.py::TestTritonTopkTopp"
|
pytest -x -v -s tests/v1/sample/test_topk_topp_sampler.py::TestTritonTopkTopp
|
||||||
|
pytest -x -v -s tests/v1/e2e/test_cpu_linear_attn_chunked_prefix.py"
|
||||||
|
|
||||||
- label: CPU-Quantization Model Tests
|
- label: CPU-Quantization Model Tests
|
||||||
depends_on: []
|
depends_on: []
|
||||||
@@ -89,11 +93,13 @@ steps:
|
|||||||
- vllm/model_executor/layers/fused_moe/experts/cpu_moe.py
|
- vllm/model_executor/layers/fused_moe/experts/cpu_moe.py
|
||||||
- tests/quantization/test_compressed_tensors.py
|
- tests/quantization/test_compressed_tensors.py
|
||||||
- tests/quantization/test_cpu_wna16.py
|
- tests/quantization/test_cpu_wna16.py
|
||||||
|
- tests/quantization/test_cpu_w8a8.py
|
||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
|
||||||
pytest -x -v -s tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_logprobs
|
pytest -x -v -s tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_logprobs
|
||||||
pytest -x -v -s tests/quantization/test_cpu_wna16.py"
|
pytest -x -v -s tests/quantization/test_cpu_wna16.py
|
||||||
|
pytest -x -v -s tests/quantization/test_cpu_w8a8.py"
|
||||||
|
|
||||||
- label: CPU-Distributed Tests (PP+TP)
|
- label: CPU-Distributed Tests (PP+TP)
|
||||||
depends_on: []
|
depends_on: []
|
||||||
@@ -135,8 +141,21 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- |
|
- |
|
||||||
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 45m "
|
||||||
pytest -x -v -s tests/models/multimodal/generation --ignore=tests/models/multimodal/generation/test_pixtral.py -m cpu_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"
|
pytest -x -v -s tests/models/multimodal/generation --ignore=tests/models/multimodal/generation/test_pixtral.py --ignore=tests/models/multimodal/generation/test_qwen2_5_vl.py -m cpu_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB"
|
||||||
parallelism: 3
|
parallelism: 4
|
||||||
|
|
||||||
|
- label: CPU-Qwen2.5-VL Multimodal Tests
|
||||||
|
depends_on: []
|
||||||
|
device: intel_cpu
|
||||||
|
no_plugin: true
|
||||||
|
source_file_dependencies:
|
||||||
|
# - vllm/
|
||||||
|
- vllm/model_executor/layers/rotary_embedding
|
||||||
|
- tests/models/multimodal/generation/
|
||||||
|
commands:
|
||||||
|
- |
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-cpu-test.sh 40m "
|
||||||
|
VLLM_CI_ENV=0 pytest -x -v -s tests/models/multimodal/generation/test_qwen2_5_vl.py"
|
||||||
|
|
||||||
- label: "Arm CPU Test"
|
- label: "Arm CPU Test"
|
||||||
depends_on: []
|
depends_on: []
|
||||||
|
|||||||
@@ -0,0 +1,80 @@
|
|||||||
|
group: Intel
|
||||||
|
steps:
|
||||||
|
- label: ":docker: Build XPU image"
|
||||||
|
soft_fail: true
|
||||||
|
optional: true
|
||||||
|
depends_on: []
|
||||||
|
key: image-build-xpu
|
||||||
|
commands:
|
||||||
|
- bash -lc '.buildkite/image_build/image_build_xpu.sh "public.ecr.aws/q9t5s3a7" "vllm-ci-test-repo" "$BUILDKITE_COMMIT"'
|
||||||
|
env:
|
||||||
|
DOCKER_BUILDKIT: "1"
|
||||||
|
retry:
|
||||||
|
automatic:
|
||||||
|
- exit_status: -1 # Agent was lost
|
||||||
|
limit: 2
|
||||||
|
- exit_status: -10 # Agent was lost
|
||||||
|
limit: 2
|
||||||
|
- label: "XPU example Test"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
optional: true
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
source_file_dependencies:
|
||||||
|
- .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
|
||||||
|
- .buildkite/scripts/hardware_ci/run-intel-ci-test.sh
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'bash .buildkite/scripts/hardware_ci/run-intel-ci-test.sh example'
|
||||||
|
- label: "XPU V1 test"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
optional: true
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
source_file_dependencies:
|
||||||
|
- .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
|
||||||
|
- .buildkite/scripts/hardware_ci/run-intel-ci-test.sh
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'bash .buildkite/scripts/hardware_ci/run-intel-ci-test.sh v1'
|
||||||
|
- label: "XPU server test"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
optional: true
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
source_file_dependencies:
|
||||||
|
- .buildkite/hardware_tests/intel_xpu_ci/test-intel.yaml
|
||||||
|
- .buildkite/scripts/hardware_ci/run-intel-ci-test.sh
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'bash .buildkite/scripts/hardware_ci/run-intel-ci-test.sh server'
|
||||||
@@ -79,12 +79,18 @@ setup_buildx_builder() {
|
|||||||
docker buildx ls | grep -E '^\*|^NAME' || docker buildx ls
|
docker buildx ls | grep -E '^\*|^NAME' || docker buildx ls
|
||||||
}
|
}
|
||||||
|
|
||||||
|
annotate_image_tags() {
|
||||||
|
.buildkite/scripts/annotate-image-build.sh \
|
||||||
|
"${IMAGE_TAG:-}" "${IMAGE_TAG_LATEST:-}"
|
||||||
|
}
|
||||||
|
|
||||||
check_and_skip_if_image_exists() {
|
check_and_skip_if_image_exists() {
|
||||||
if [[ -n "${IMAGE_TAG:-}" ]]; then
|
if [[ -n "${IMAGE_TAG:-}" ]]; then
|
||||||
echo "--- :mag: Checking if image exists"
|
echo "--- :mag: Checking if image exists"
|
||||||
if docker manifest inspect "${IMAGE_TAG}" >/dev/null 2>&1; then
|
if docker manifest inspect "${IMAGE_TAG}" >/dev/null 2>&1; then
|
||||||
echo "Image already exists: ${IMAGE_TAG}"
|
echo "Image already exists: ${IMAGE_TAG}"
|
||||||
echo "Skipping build"
|
echo "Skipping build"
|
||||||
|
annotate_image_tags
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
echo "Image not found, proceeding with build"
|
echo "Image not found, proceeding with build"
|
||||||
@@ -171,6 +177,18 @@ BRANCH=$4
|
|||||||
IMAGE_TAG=$5
|
IMAGE_TAG=$5
|
||||||
IMAGE_TAG_LATEST=${6:-} # only used for main branch, optional
|
IMAGE_TAG_LATEST=${6:-} # only used for main branch, optional
|
||||||
|
|
||||||
|
# When TORCH_NIGHTLY=1, build the base CI image against PyTorch nightly so the
|
||||||
|
# entire existing pipeline runs on nightly torch (CUDA/GPU lane only). Delegate
|
||||||
|
# to the dedicated nightly build (PYTORCH_NIGHTLY=1, CUDA 13.0) and tag it at the
|
||||||
|
# normal IMAGE_TAG that every test step already pulls -- no separate image tag,
|
||||||
|
# no duplicate "vLLM Against PyTorch Nightly" pipeline section.
|
||||||
|
if [[ "${TORCH_NIGHTLY:-0}" == "1" ]]; then
|
||||||
|
echo "--- :warning: TORCH_NIGHTLY=1 -- building base image on PyTorch nightly"
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
exec "${SCRIPT_DIR}/image_build_torch_nightly.sh" \
|
||||||
|
"${REGISTRY}" "${REPO}" "${BUILDKITE_COMMIT}" "${BRANCH}" "${IMAGE_TAG}"
|
||||||
|
fi
|
||||||
|
|
||||||
# build config
|
# build config
|
||||||
TARGET="test-ci"
|
TARGET="test-ci"
|
||||||
VLLM_BAKE_FILE_PATH="${VLLM_BAKE_FILE_PATH:-docker/docker-bake.hcl}"
|
VLLM_BAKE_FILE_PATH="${VLLM_BAKE_FILE_PATH:-docker/docker-bake.hcl}"
|
||||||
@@ -254,3 +272,5 @@ echo "--- :docker: Building ${TARGET}"
|
|||||||
docker --debug buildx bake -f "${VLLM_BAKE_FILE_PATH}" -f "${CI_HCL_PATH}" --progress plain "${TARGET}"
|
docker --debug buildx bake -f "${VLLM_BAKE_FILE_PATH}" -f "${CI_HCL_PATH}" --progress plain "${TARGET}"
|
||||||
|
|
||||||
echo "--- :white_check_mark: Build complete"
|
echo "--- :white_check_mark: Build complete"
|
||||||
|
|
||||||
|
annotate_image_tags
|
||||||
|
|||||||
@@ -9,29 +9,31 @@ fi
|
|||||||
REGISTRY=$1
|
REGISTRY=$1
|
||||||
REPO=$2
|
REPO=$2
|
||||||
BUILDKITE_COMMIT=$3
|
BUILDKITE_COMMIT=$3
|
||||||
|
IMAGE="$REGISTRY/$REPO:$BUILDKITE_COMMIT-arm64"
|
||||||
|
|
||||||
# authenticate with AWS ECR
|
# authenticate with AWS ECR
|
||||||
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
||||||
|
|
||||||
# skip build if image already exists
|
# skip build if image already exists
|
||||||
if [[ -z $(docker manifest inspect "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64) ]]; then
|
if docker manifest inspect "$IMAGE" >/dev/null 2>&1; then
|
||||||
echo "Image not found, proceeding with build..."
|
|
||||||
else
|
|
||||||
echo "Image found"
|
echo "Image found"
|
||||||
exit 0
|
else
|
||||||
|
echo "Image not found, proceeding with build..."
|
||||||
|
# build for arm64 GPU targets: Grace/GH200 (sm_90),
|
||||||
|
# Blackwell/Thor (sm_100/sm_103/sm_110), and DGX Spark/GB10
|
||||||
|
# (sm_121, family-covered by 12.0 under CUDA 13)
|
||||||
|
docker build --file docker/Dockerfile \
|
||||||
|
--platform linux/arm64 \
|
||||||
|
--build-arg max_jobs=16 \
|
||||||
|
--build-arg nvcc_threads=4 \
|
||||||
|
--build-arg torch_cuda_arch_list="9.0 10.0 11.0 12.0" \
|
||||||
|
--build-arg USE_SCCACHE=1 \
|
||||||
|
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
||||||
|
--tag "$IMAGE" \
|
||||||
|
--target test \
|
||||||
|
--progress plain .
|
||||||
|
# push
|
||||||
|
docker push "$IMAGE"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# build (Grace/GH200 is the arm64 GPU target; sm_90)
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE"
|
||||||
docker build --file docker/Dockerfile \
|
|
||||||
--platform linux/arm64 \
|
|
||||||
--build-arg max_jobs=16 \
|
|
||||||
--build-arg nvcc_threads=4 \
|
|
||||||
--build-arg torch_cuda_arch_list="9.0" \
|
|
||||||
--build-arg USE_SCCACHE=1 \
|
|
||||||
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
|
||||||
--tag "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64 \
|
|
||||||
--target test \
|
|
||||||
--progress plain .
|
|
||||||
|
|
||||||
# push
|
|
||||||
docker push "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64
|
|
||||||
|
|||||||
@@ -9,26 +9,26 @@ fi
|
|||||||
REGISTRY=$1
|
REGISTRY=$1
|
||||||
REPO=$2
|
REPO=$2
|
||||||
BUILDKITE_COMMIT=$3
|
BUILDKITE_COMMIT=$3
|
||||||
|
IMAGE="$REGISTRY/$REPO:$BUILDKITE_COMMIT-cpu"
|
||||||
|
|
||||||
# authenticate with AWS ECR
|
# authenticate with AWS ECR
|
||||||
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
||||||
|
|
||||||
# skip build if image already exists
|
# skip build if image already exists
|
||||||
if [[ -z $(docker manifest inspect "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-cpu) ]]; then
|
if docker manifest inspect "$IMAGE" >/dev/null 2>&1; then
|
||||||
echo "Image not found, proceeding with build..."
|
|
||||||
else
|
|
||||||
echo "Image found"
|
echo "Image found"
|
||||||
exit 0
|
else
|
||||||
|
echo "Image not found, proceeding with build..."
|
||||||
|
# build
|
||||||
|
docker build --file docker/Dockerfile.cpu \
|
||||||
|
--build-arg max_jobs=16 \
|
||||||
|
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
||||||
|
--build-arg VLLM_CPU_X86=true \
|
||||||
|
--tag "$IMAGE" \
|
||||||
|
--target vllm-test \
|
||||||
|
--progress plain .
|
||||||
|
# push
|
||||||
|
docker push "$IMAGE"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# build
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE"
|
||||||
docker build --file docker/Dockerfile.cpu \
|
|
||||||
--build-arg max_jobs=16 \
|
|
||||||
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
|
||||||
--build-arg VLLM_CPU_X86=true \
|
|
||||||
--tag "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-cpu \
|
|
||||||
--target vllm-test \
|
|
||||||
--progress plain .
|
|
||||||
|
|
||||||
# push
|
|
||||||
docker push "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-cpu
|
|
||||||
|
|||||||
@@ -9,25 +9,25 @@ fi
|
|||||||
REGISTRY=$1
|
REGISTRY=$1
|
||||||
REPO=$2
|
REPO=$2
|
||||||
BUILDKITE_COMMIT=$3
|
BUILDKITE_COMMIT=$3
|
||||||
|
IMAGE="$REGISTRY/$REPO:$BUILDKITE_COMMIT-arm64-cpu"
|
||||||
|
|
||||||
# authenticate with AWS ECR
|
# authenticate with AWS ECR
|
||||||
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
||||||
|
|
||||||
# skip build if image already exists
|
# skip build if image already exists
|
||||||
if [[ -z $(docker manifest inspect "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64-cpu) ]]; then
|
if docker manifest inspect "$IMAGE" >/dev/null 2>&1; then
|
||||||
echo "Image not found, proceeding with build..."
|
|
||||||
else
|
|
||||||
echo "Image found"
|
echo "Image found"
|
||||||
exit 0
|
else
|
||||||
|
echo "Image not found, proceeding with build..."
|
||||||
|
# build
|
||||||
|
docker build --file docker/Dockerfile.cpu \
|
||||||
|
--build-arg max_jobs=16 \
|
||||||
|
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
||||||
|
--tag "$IMAGE" \
|
||||||
|
--target vllm-test \
|
||||||
|
--progress plain .
|
||||||
|
# push
|
||||||
|
docker push "$IMAGE"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# build
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE"
|
||||||
docker build --file docker/Dockerfile.cpu \
|
|
||||||
--build-arg max_jobs=16 \
|
|
||||||
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
|
||||||
--tag "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64-cpu \
|
|
||||||
--target vllm-test \
|
|
||||||
--progress plain .
|
|
||||||
|
|
||||||
# push
|
|
||||||
docker push "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-arm64-cpu
|
|
||||||
|
|||||||
@@ -9,26 +9,26 @@ fi
|
|||||||
REGISTRY=$1
|
REGISTRY=$1
|
||||||
REPO=$2
|
REPO=$2
|
||||||
BUILDKITE_COMMIT=$3
|
BUILDKITE_COMMIT=$3
|
||||||
|
IMAGE="$REGISTRY/$REPO:$BUILDKITE_COMMIT-hpu"
|
||||||
|
|
||||||
# authenticate with AWS ECR
|
# authenticate with AWS ECR
|
||||||
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
||||||
|
|
||||||
# skip build if image already exists
|
# skip build if image already exists
|
||||||
if [[ -z $(docker manifest inspect "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-hpu) ]]; then
|
if docker manifest inspect "$IMAGE" >/dev/null 2>&1; then
|
||||||
echo "Image not found, proceeding with build..."
|
|
||||||
else
|
|
||||||
echo "Image found"
|
echo "Image found"
|
||||||
exit 0
|
else
|
||||||
|
echo "Image not found, proceeding with build..."
|
||||||
|
# build
|
||||||
|
docker build \
|
||||||
|
--file tests/pytorch_ci_hud_benchmark/Dockerfile.hpu \
|
||||||
|
--build-arg max_jobs=16 \
|
||||||
|
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
||||||
|
--tag "$IMAGE" \
|
||||||
|
--progress plain \
|
||||||
|
https://github.com/vllm-project/vllm-gaudi.git
|
||||||
|
# push
|
||||||
|
docker push "$IMAGE"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# build
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE"
|
||||||
docker build \
|
|
||||||
--file tests/pytorch_ci_hud_benchmark/Dockerfile.hpu \
|
|
||||||
--build-arg max_jobs=16 \
|
|
||||||
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
|
||||||
--tag "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-hpu \
|
|
||||||
--progress plain \
|
|
||||||
https://github.com/vllm-project/vllm-gaudi.git
|
|
||||||
|
|
||||||
# push
|
|
||||||
docker push "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-hpu
|
|
||||||
|
|||||||
@@ -40,6 +40,7 @@ docker buildx ls
|
|||||||
echo "--- :mag: Checking if image already exists"
|
echo "--- :mag: Checking if image already exists"
|
||||||
if docker manifest inspect "$IMAGE_TAG" >/dev/null 2>&1; then
|
if docker manifest inspect "$IMAGE_TAG" >/dev/null 2>&1; then
|
||||||
echo "Image found: $IMAGE_TAG — skipping build"
|
echo "Image found: $IMAGE_TAG — skipping build"
|
||||||
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE_TAG"
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
echo "Image not found, proceeding with build..."
|
echo "Image not found, proceeding with build..."
|
||||||
@@ -66,3 +67,5 @@ docker buildx build --file docker/Dockerfile \
|
|||||||
--progress plain .
|
--progress plain .
|
||||||
|
|
||||||
echo "--- :white_check_mark: Torch nightly image build complete: $IMAGE_TAG"
|
echo "--- :white_check_mark: Torch nightly image build complete: $IMAGE_TAG"
|
||||||
|
|
||||||
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE_TAG"
|
||||||
|
|||||||
@@ -9,26 +9,26 @@ fi
|
|||||||
REGISTRY=$1
|
REGISTRY=$1
|
||||||
REPO=$2
|
REPO=$2
|
||||||
BUILDKITE_COMMIT=$3
|
BUILDKITE_COMMIT=$3
|
||||||
|
IMAGE="$REGISTRY/$REPO:$BUILDKITE_COMMIT-xpu"
|
||||||
|
|
||||||
# authenticate with AWS ECR
|
# authenticate with AWS ECR
|
||||||
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin "$REGISTRY" || true
|
||||||
aws ecr get-login-password --region us-east-1 | docker login --username AWS --password-stdin 936637512419.dkr.ecr.us-east-1.amazonaws.com || true
|
aws ecr get-login-password --region us-east-1 | docker login --username AWS --password-stdin 936637512419.dkr.ecr.us-east-1.amazonaws.com || true
|
||||||
|
|
||||||
# skip build if image already exists
|
# skip build if image already exists
|
||||||
if ! docker manifest inspect "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-xpu &> /dev/null; then
|
if docker manifest inspect "$IMAGE" &> /dev/null; then
|
||||||
echo "Image not found, proceeding with build..."
|
|
||||||
else
|
|
||||||
echo "Image found"
|
echo "Image found"
|
||||||
exit 0
|
else
|
||||||
|
echo "Image not found, proceeding with build..."
|
||||||
|
# build
|
||||||
|
docker build \
|
||||||
|
--file docker/Dockerfile.xpu \
|
||||||
|
--build-arg max_jobs=16 \
|
||||||
|
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
||||||
|
--tag "$IMAGE" \
|
||||||
|
--progress plain .
|
||||||
|
# push
|
||||||
|
docker push "$IMAGE"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# build
|
.buildkite/scripts/annotate-image-build.sh "$IMAGE"
|
||||||
docker build \
|
|
||||||
--file docker/Dockerfile.xpu \
|
|
||||||
--build-arg max_jobs=16 \
|
|
||||||
--build-arg buildkite_commit="$BUILDKITE_COMMIT" \
|
|
||||||
--tag "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-xpu \
|
|
||||||
--progress plain .
|
|
||||||
|
|
||||||
# push
|
|
||||||
docker push "$REGISTRY"/"$REPO":"$BUILDKITE_COMMIT"-xpu
|
|
||||||
|
|||||||
@@ -5,6 +5,10 @@ steps:
|
|||||||
- label: XPU Sleep Mode
|
- label: XPU Sleep Mode
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -19,4 +23,5 @@ steps:
|
|||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'cd tests &&
|
'cd tests &&
|
||||||
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
||||||
|
pytest -v -s basic_correctness/test_cpu_offload.py &&
|
||||||
pytest -v -s basic_correctness/test_mem.py::test_end_to_end'
|
pytest -v -s basic_correctness/test_mem.py::test_end_to_end'
|
||||||
|
|||||||
@@ -5,6 +5,10 @@ steps:
|
|||||||
- label: Engine (1 GPU)
|
- label: Engine (1 GPU)
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -1,11 +1,15 @@
|
|||||||
group: Expert Parallelism
|
group: Expert Parallelism
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
steps:
|
steps:
|
||||||
- label: EPLB Algorithm
|
- label: EPLB Algorithm
|
||||||
key: eplb-algorithm
|
key: eplb-algorithm
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -5,6 +5,10 @@ steps:
|
|||||||
- label: vLLM IR Tests
|
- label: vLLM IR Tests
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -5,6 +5,10 @@ steps:
|
|||||||
- label: LoRA Runtime + Utils
|
- label: LoRA Runtime + Utils
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -34,6 +38,10 @@ steps:
|
|||||||
- label: LoRA Fused/MoE Kernels
|
- label: LoRA Fused/MoE Kernels
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -54,6 +62,10 @@ steps:
|
|||||||
- label: LoRA Punica Kernels
|
- label: LoRA Punica Kernels
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -69,11 +81,17 @@ steps:
|
|||||||
'cd tests &&
|
'cd tests &&
|
||||||
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
||||||
set -o pipefail &&
|
set -o pipefail &&
|
||||||
pytest -v -s lora/test_punica_ops.py --deselect="tests/lora/test_punica_ops.py::test_kernels_hidden_size[expand-0-xpu:0-dtype0-3-43264-32-4-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels[shrink-0-xpu:0-dtype1-1-2049-64-128-16]" --deselect="tests/lora/test_punica_ops.py::test_kernels[shrink-0-xpu:0-dtype0-1-2049-128-1-32]" --deselect="tests/lora/test_punica_ops.py::test_kernels[shrink-0-xpu:0-dtype0-1-2049-256-1-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels[shrink-0-xpu:0-dtype0-1-2049-256-8-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels[expand-0-xpu:0-dtype0-3-2049-128-8-16]" --deselect="tests/lora/test_punica_ops.py::test_kernels[shrink-0-xpu:0-dtype0-1-2049-128-8-32]" --deselect="tests/lora/test_punica_ops.py::test_kernels[expand-0-xpu:0-dtype1-1-2049-256-128-32]" --deselect="tests/lora/test_punica_ops.py::test_kernels_hidden_size[shrink-0-xpu:0-dtype0-3-64256-32-4-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels_hidden_size[shrink-0-xpu:0-dtype1-2-29696-32-4-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels_hidden_size[shrink-0-xpu:0-dtype1-3-49408-32-4-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels_hidden_size[shrink-0-xpu:0-dtype0-2-16384-32-4-4]" --deselect="tests/lora/test_punica_ops.py::test_kernels_hidden_size[expand-0-xpu:0-dtype0-2-51328-32-4-4]"'
|
pytest -v -s lora/test_punica_ops.py::test_kernels &&
|
||||||
|
pytest -v -s lora/test_punica_ops.py::test_kernels_hidden_size &&
|
||||||
|
pytest -v -s lora/test_punica_ops.py::test_add_lora_fused_moe_early_exit'
|
||||||
|
|
||||||
- label: LoRA Punica FP8/XPU Ops
|
- label: LoRA Punica FP8/XPU Ops
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -94,6 +112,10 @@ steps:
|
|||||||
- label: LoRA Models
|
- label: LoRA Models
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -108,15 +130,19 @@ steps:
|
|||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'cd tests &&
|
'cd tests &&
|
||||||
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
||||||
(pytest -v -s lora/test_mixtral.py --deselect="tests/lora/test_mixtral.py::test_mixtral_lora[4]" || true) &&
|
|
||||||
pytest -v -s lora/test_quant_model.py --deselect="tests/lora/test_quant_model.py::test_quant_model_lora[model0]" --deselect="tests/lora/test_quant_model.py::test_quant_model_lora[model1]" --deselect="tests/lora/test_quant_model.py::test_quant_model_tp_equality[model0]" &&
|
pytest -v -s lora/test_quant_model.py --deselect="tests/lora/test_quant_model.py::test_quant_model_lora[model0]" --deselect="tests/lora/test_quant_model.py::test_quant_model_lora[model1]" --deselect="tests/lora/test_quant_model.py::test_quant_model_tp_equality[model0]" &&
|
||||||
pytest -v -s lora/test_transformers_model.py &&
|
pytest -v -s lora/test_transformers_model.py &&
|
||||||
pytest -v -s lora/test_chatglm3_tp.py &&
|
pytest -v -s lora/test_chatglm3_tp.py &&
|
||||||
|
pytest -v -s lora/test_llama_tp.py::test_llama_lora &&
|
||||||
pytest -s -v lora/test_minicpmv_tp.py'
|
pytest -s -v lora/test_minicpmv_tp.py'
|
||||||
|
|
||||||
- label: LoRA Multimodal
|
- label: LoRA Multimodal
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -5,6 +5,10 @@ steps:
|
|||||||
- label: V1 Core + KV + Metrics
|
- label: V1 Core + KV + Metrics
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -31,6 +35,10 @@ steps:
|
|||||||
- label: V1 Sample + Logits
|
- label: V1 Sample + Logits
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -57,17 +65,46 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
'pip install lm_eval[api]>=0.4.12 &&
|
||||||
|
export VLLM_WORKER_MULTIPROC_METHOD=spawn &&
|
||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s v1/logits_processors --ignore=v1/logits_processors/test_custom_online.py --ignore=v1/logits_processors/test_custom_offline.py &&
|
pytest -v -s v1/logits_processors --ignore=v1/logits_processors/test_custom_online.py --ignore=v1/logits_processors/test_custom_offline.py &&
|
||||||
pytest -v -s v1/test_oracle.py &&
|
pytest -v -s v1/test_oracle.py &&
|
||||||
pytest -v -s v1/test_request.py &&
|
pytest -v -s v1/test_request.py &&
|
||||||
pytest -v -s v1/test_outputs.py &&
|
pytest -v -s v1/test_outputs.py &&
|
||||||
pytest -v -s v1/sample/test_topk_topp_sampler.py'
|
pytest -v -s v1/sample'
|
||||||
|
|
||||||
|
- label: Basic Models Tests (Initialization)
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- tests/models/test_initialization.py
|
||||||
|
- tests/models/registry.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'export VLLM_XPU_FUSED_MOE_USE_REF=1 &&
|
||||||
|
cd tests &&
|
||||||
|
pytest -v -s models/test_initialization.py::test_can_initialize_large_subset[Eagle3MiniMaxM2ForCausalLM]'
|
||||||
|
|
||||||
- label: XPU CPU Offload
|
- label: XPU CPU Offload
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 60
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -88,10 +125,39 @@ steps:
|
|||||||
pytest -v -s v1/kv_offload &&
|
pytest -v -s v1/kv_offload &&
|
||||||
pytest -v -s v1/kv_connector/unit/test_offloading_connector.py'
|
pytest -v -s v1/kv_connector/unit/test_offloading_connector.py'
|
||||||
|
|
||||||
|
- label: NixlConnector PD accuracy (2 GPUs)
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
num_devices: 2
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- vllm/v1/worker/kv_connector_model_runner_mixin.py
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- vllm/platforms/xpu.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
bash v1/kv_connector/nixl_integration/run_xpu_disagg_accuracy_test.sh'
|
||||||
|
|
||||||
- label: Regression
|
- label: Regression
|
||||||
key: regression
|
key: regression
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -114,7 +180,7 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'pip install modelscope &&
|
'pip install modelscope\<1.38 &&
|
||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s test_regression.py'
|
pytest -v -s test_regression.py'
|
||||||
|
|
||||||
@@ -123,6 +189,10 @@ steps:
|
|||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -154,6 +224,10 @@ steps:
|
|||||||
key: async-engine-inputs-utils-worker
|
key: async-engine-inputs-utils-worker
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -0,0 +1,62 @@
|
|||||||
|
group: Model Runner V2 Intel
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
steps:
|
||||||
|
- label: Model Runner V2 Core Tests (Intel)
|
||||||
|
timeout_in_minutes: 45
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/worker/gpu/
|
||||||
|
- vllm/v1/worker/gpu_worker.py
|
||||||
|
- vllm/v1/core/sched/
|
||||||
|
- vllm/v1/attention/
|
||||||
|
- tests/v1/engine/test_llm_engine.py
|
||||||
|
- tests/v1/e2e/
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'export VLLM_USE_V2_MODEL_RUNNER=1 &&
|
||||||
|
cd tests &&
|
||||||
|
pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics" &&
|
||||||
|
ENFORCE_EAGER=1 pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram" &&
|
||||||
|
pytest -v -s v1/e2e/general/test_min_tokens.py'
|
||||||
|
|
||||||
|
- label: Model Runner V2 Examples (Intel)
|
||||||
|
timeout_in_minutes: 45
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/worker/gpu/
|
||||||
|
- vllm/v1/core/sched/
|
||||||
|
- vllm/v1/worker/gpu_worker.py
|
||||||
|
- examples/basic/offline_inference/
|
||||||
|
- examples/generate/multimodal/
|
||||||
|
- examples/features/
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'export VLLM_USE_V2_MODEL_RUNNER=1 &&
|
||||||
|
cd examples &&
|
||||||
|
python3 basic/offline_inference/chat.py &&
|
||||||
|
python3 basic/offline_inference/generate.py --model facebook/opt-125m &&
|
||||||
|
python3 generate/multimodal/vision_language_offline.py --seed 0 &&
|
||||||
|
python3 features/automatic_prefix_caching/prefix_caching_offline.py'
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
group: Models - Distributed
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
steps:
|
||||||
|
- label: Distributed Model Tests (2 GPUs)
|
||||||
|
key: distributed-model-tests-2-gpus
|
||||||
|
timeout_in_minutes: 50
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/model_loader/sharded_state_loader.py
|
||||||
|
- vllm/model_executor/models/
|
||||||
|
- tests/model_executor/model_loader/test_sharded_state_loader.py
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s model_executor/model_loader/test_sharded_state_loader.py -m "not slow_test"'
|
||||||
@@ -1,11 +1,15 @@
|
|||||||
group: Models - Multimodal
|
group: Models - Multimodal
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
steps:
|
steps:
|
||||||
- label: "Multi-Modal Models (Standard) 1: qwen2"
|
- label: "Multi-Modal Models (Standard) 1: qwen2"
|
||||||
key: multi-modal-models-standard-1-qwen2
|
key: multi-modal-models-standard-1-qwen2
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -18,7 +22,7 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'pip install av git+https://github.com/TIGER-AI-Lab/Mantis.git &&
|
'pip install av &&
|
||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen2" &&
|
pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen2" &&
|
||||||
pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model'
|
pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model'
|
||||||
@@ -27,6 +31,10 @@ steps:
|
|||||||
key: multi-modal-models-standard-2-qwen3-gemma
|
key: multi-modal-models-standard-2-qwen3-gemma
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -39,14 +47,17 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'pip install git+https://github.com/TIGER-AI-Lab/Mantis.git &&
|
'cd tests &&
|
||||||
cd tests &&
|
|
||||||
pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model'
|
pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model'
|
||||||
|
|
||||||
- label: "Multi-Modal Models (Standard) 3: llava + qwen2_vl"
|
- label: "Multi-Modal Models (Standard) 3: llava + qwen2_vl"
|
||||||
key: multi-modal-models-standard-3-llava-qwen2-vl
|
key: multi-modal-models-standard-3-llava-qwen2-vl
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -59,8 +70,7 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'pip install git+https://github.com/TIGER-AI-Lab/Mantis.git &&
|
'cd tests &&
|
||||||
cd tests &&
|
|
||||||
pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "not qwen2 and not qwen3 and not gemma" &&
|
pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "not qwen2 and not qwen3 and not gemma" &&
|
||||||
pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model'
|
pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model'
|
||||||
|
|
||||||
@@ -68,6 +78,10 @@ steps:
|
|||||||
key: multi-modal-models-standard-4-other-whisper
|
key: multi-modal-models-standard-4-other-whisper
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -80,7 +94,7 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'pip install av git+https://github.com/TIGER-AI-Lab/Mantis.git &&
|
'pip install av &&
|
||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/processing'
|
pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/processing'
|
||||||
|
|
||||||
@@ -88,6 +102,10 @@ steps:
|
|||||||
key: multi-modal-processor
|
key: multi-modal-processor
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 45
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
working_dir: "."
|
working_dir: "."
|
||||||
env:
|
env:
|
||||||
@@ -101,11 +119,9 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- >-
|
- >-
|
||||||
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
'pip install av matplotlib ftfy git+https://github.com/TIGER-AI-Lab/Mantis.git &&
|
'pip install av matplotlib ftfy &&
|
||||||
pip install open-clip-torch --no-deps &&
|
pip install open-clip-torch --no-deps &&
|
||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s models/multimodal/processing/test_tensor_schema.py
|
pytest -v -s models/multimodal/processing/test_tensor_schema.py
|
||||||
--deselect "tests/models/multimodal/processing/test_tensor_schema.py::test_model_tensor_schema[mistralai/Mistral-Large-3-675B-Instruct-2512-NVFP4]"
|
|
||||||
--deselect "tests/models/multimodal/processing/test_tensor_schema.py::test_model_tensor_schema[Qwen/Qwen2.5-Omni-7B-AWQ]"
|
|
||||||
--num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB'
|
--num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB'
|
||||||
parallelism: 4
|
parallelism: 4
|
||||||
|
|||||||
@@ -0,0 +1,28 @@
|
|||||||
|
group: Quantization
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
steps:
|
||||||
|
- label: Quantization
|
||||||
|
key: quantization
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
no_plugin: true
|
||||||
|
working_dir: "."
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/
|
||||||
|
- vllm/model_executor/layers/quantization
|
||||||
|
- tests/quantization
|
||||||
|
commands:
|
||||||
|
# - VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s quantization/ --ignore quantization/test_blackwell_moe.py
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'VLLM_TEST_FORCE_LOAD_FORMAT=auto pytest -v -s tests/quantization/test_per_token_kv_cache.py --deselect="tests/quantization/test_per_token_kv_cache.py::test_triton_unified_attention_per_token_head_scale[int4-16-128-num_heads0-seq_lens1]"'
|
||||||
|
|
||||||
@@ -19,6 +19,10 @@ steps:
|
|||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 2+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
env:
|
env:
|
||||||
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
@@ -38,17 +42,46 @@ steps:
|
|||||||
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager --kv-cache-dtype fp8 &&
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager --kv-cache-dtype fp8 &&
|
||||||
python3 examples/basic/offline_inference/generate.py --model nvidia/Llama-3.1-8B-Instruct-FP8 --block-size 64 --enforce-eager --quantization modelopt --kv-cache-dtype fp8 --attention-backend TRITON_ATTN --max-model-len 4096 &&
|
python3 examples/basic/offline_inference/generate.py --model nvidia/Llama-3.1-8B-Instruct-FP8 --block-size 64 --enforce-eager --quantization modelopt --kv-cache-dtype fp8 --attention-backend TRITON_ATTN --max-model-len 4096 &&
|
||||||
python3 examples/basic/offline_inference/generate.py --model superjob/Qwen3-4B-Instruct-2507-GPTQ-Int4 --block-size 64 --enforce-eager --max-model-len 8192 &&
|
python3 examples/basic/offline_inference/generate.py --model superjob/Qwen3-4B-Instruct-2507-GPTQ-Int4 --block-size 64 --enforce-eager --max-model-len 8192 &&
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model TheBloke/TinyLlama-1.1B-Chat-v0.3-AWQ --block-size 64 --enforce-eager &&
|
||||||
python3 examples/basic/offline_inference/generate.py --model ibm-research/PowerMoE-3b --block-size 64 --enforce-eager -tp 2 &&
|
python3 examples/basic/offline_inference/generate.py --model ibm-research/PowerMoE-3b --block-size 64 --enforce-eager -tp 2 &&
|
||||||
python3 examples/basic/offline_inference/generate.py --model ibm-research/PowerMoE-3b --block-size 64 --enforce-eager -tp 2 --enable-expert-parallel &&
|
python3 examples/basic/offline_inference/generate.py --model ibm-research/PowerMoE-3b --block-size 64 --enforce-eager -tp 2 --enable-expert-parallel &&
|
||||||
python3 examples/basic/offline_inference/generate.py --model superjob/Qwen3-4B-Instruct-2507-GPTQ-Int4 --max-model-len 8192 &&
|
python3 examples/basic/offline_inference/generate.py --model superjob/Qwen3-4B-Instruct-2507-GPTQ-Int4 --max-model-len 8192 &&
|
||||||
VLLM_XPU_FUSED_MOE_USE_REF=1 python3 examples/basic/offline_inference/generate.py --model Qwen/Qwen3-30B-A3B-Instruct-2507-FP8 --enforce-eager -tp 2 --max-model-len 8192 &&
|
VLLM_XPU_FUSED_MOE_USE_REF=1 python3 examples/basic/offline_inference/generate.py --model Qwen/Qwen3-30B-A3B-Instruct-2507-FP8 --enforce-eager -tp 2 --max-model-len 8192 &&
|
||||||
python3 examples/basic/offline_inference/generate.py --model INCModel/Qwen3-30B-A3B-Instruct-2507-MXFP4-LLMC --enforce-eager -tp 2 --max-model-len 8192
|
python3 examples/basic/offline_inference/generate.py --model INCModel/Qwen3-30B-A3B-Instruct-2507-MXFP4-LLMC --enforce-eager -tp 2 --max-model-len 8192
|
||||||
'
|
'
|
||||||
|
- label: "XPU W8A8 FP8 Linear Examples"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- .buildkite/intel_jobs/test-intel.yaml
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'python3 examples/basic/offline_inference/generate.py --linear-backend xpu --model RedHatAI/Meta-Llama-3.1-8B-Instruct-FP8 --enforce-eager --max-model-len 4096 &&
|
||||||
|
python3 examples/basic/offline_inference/generate.py --linear-backend xpu --model neuralmagic/Llama-3.2-1B-Instruct-FP8-dynamic --enforce-eager --max-model-len 4096 &&
|
||||||
|
python3 examples/basic/offline_inference/generate.py --linear-backend xpu --model meta-llama/Llama-3.2-1B-Instruct --quantization fp8 --enforce-eager --max-model-len 4096
|
||||||
|
'
|
||||||
- label: "XPU V1 test"
|
- label: "XPU V1 test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 24+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
env:
|
env:
|
||||||
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
@@ -66,13 +99,17 @@ steps:
|
|||||||
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py &&
|
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py &&
|
||||||
pytest -v -s v1/structured_output &&
|
pytest -v -s v1/structured_output &&
|
||||||
pytest -v -s v1/test_serial_utils.py &&
|
pytest -v -s v1/test_serial_utils.py &&
|
||||||
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py &&
|
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py --ignore=v1/spec_decode/test_speculators_correctness.py &&
|
||||||
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py'
|
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py'
|
||||||
- label: "XPU server test"
|
- label: "XPU server test"
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-xpu
|
- image-build-xpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
device: intel_gpu
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
env:
|
env:
|
||||||
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
@@ -87,3 +124,47 @@ steps:
|
|||||||
cd tests &&
|
cd tests &&
|
||||||
pytest -v -s entrypoints/multimodal/openai/chat_completion/test_audio_in_video.py &&
|
pytest -v -s entrypoints/multimodal/openai/chat_completion/test_audio_in_video.py &&
|
||||||
pytest -v -s benchmarks/test_serve_cli.py'
|
pytest -v -s benchmarks/test_serve_cli.py'
|
||||||
|
- label: "XPU quantization test"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- .buildkite/intel_jobs/test-intel.yaml
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s quantization/test_auto_round.py'
|
||||||
|
- label: "XPU compressed tensors FP8 test"
|
||||||
|
depends_on:
|
||||||
|
- image-build-xpu
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: intel_gpu
|
||||||
|
agent_tags:
|
||||||
|
label: production
|
||||||
|
gpu: 1+
|
||||||
|
mem: 16+
|
||||||
|
no_plugin: true
|
||||||
|
env:
|
||||||
|
REGISTRY: "public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO: "vllm-ci-test-repo"
|
||||||
|
VLLM_TEST_DEVICE: "xpu"
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/
|
||||||
|
- tests/quantization/test_compressed_tensors.py
|
||||||
|
- .buildkite/intel_jobs/test-intel.yaml
|
||||||
|
commands:
|
||||||
|
- >-
|
||||||
|
bash .buildkite/scripts/hardware_ci/run-intel-test.sh
|
||||||
|
'cd tests &&
|
||||||
|
pytest -v -s quantization/test_compressed_tensors.py::test_compressed_tensors_fp8'
|
||||||
@@ -350,6 +350,25 @@ steps:
|
|||||||
- "docker push public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-$(uname -m)-cu129-ubuntu2404"
|
- "docker push public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-$(uname -m)-cu129-ubuntu2404"
|
||||||
- 'bash .buildkite/scripts/annotate-build-artifact.sh "$$BUILDKITE_LABEL" "public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-$(uname -m)-cu129-ubuntu2404"'
|
- 'bash .buildkite/scripts/annotate-build-artifact.sh "$$BUILDKITE_LABEL" "public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-$(uname -m)-cu129-ubuntu2404"'
|
||||||
|
|
||||||
|
- label: ":docker: Build release image - x86_64 - XPU"
|
||||||
|
depends_on: ~
|
||||||
|
id: build-xpu-release-image
|
||||||
|
agents:
|
||||||
|
queue: cpu_queue_release
|
||||||
|
commands:
|
||||||
|
- "aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin public.ecr.aws/q9t5s3a7"
|
||||||
|
- |
|
||||||
|
DOCKER_BUILDKIT=1 docker build \
|
||||||
|
$(bash .buildkite/scripts/docker-build-metadata-args.sh xpu) \
|
||||||
|
--build-arg GIT_REPO_CHECK=1 \
|
||||||
|
--target vllm-openai \
|
||||||
|
--progress plain \
|
||||||
|
-f docker/Dockerfile.xpu .
|
||||||
|
- "docker push public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-$(uname -m)-xpu"
|
||||||
|
- 'bash .buildkite/scripts/annotate-build-artifact.sh "$$BUILDKITE_LABEL" "public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-$(uname -m)-xpu"'
|
||||||
|
env:
|
||||||
|
DOCKER_BUILDKIT: "1"
|
||||||
|
|
||||||
- block: "Build release image for x86_64 CPU"
|
- block: "Build release image for x86_64 CPU"
|
||||||
key: block-cpu-release-image-build
|
key: block-cpu-release-image-build
|
||||||
depends_on: ~
|
depends_on: ~
|
||||||
@@ -445,6 +464,16 @@ steps:
|
|||||||
- "docker manifest push public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-cu129-ubuntu2404"
|
- "docker manifest push public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-cu129-ubuntu2404"
|
||||||
- 'bash .buildkite/scripts/annotate-build-artifact.sh "Manifest: CUDA 12.9 Ubuntu 24.04" "public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-cu129-ubuntu2404"'
|
- 'bash .buildkite/scripts/annotate-build-artifact.sh "Manifest: CUDA 12.9 Ubuntu 24.04" "public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-cu129-ubuntu2404"'
|
||||||
|
|
||||||
|
- label: "Create manifest - XPU"
|
||||||
|
depends_on:
|
||||||
|
- build-xpu-release-image
|
||||||
|
id: create-manifest-xpu
|
||||||
|
agents:
|
||||||
|
queue: small_cpu_queue_release
|
||||||
|
commands:
|
||||||
|
- "bash .buildkite/scripts/xpu/create-xpu-ecr-manifest.sh"
|
||||||
|
- 'bash .buildkite/scripts/annotate-build-artifact.sh "Manifest: XPU" "public.ecr.aws/q9t5s3a7/vllm-release-repo:$BUILDKITE_COMMIT-xpu"'
|
||||||
|
|
||||||
- label: "Publish nightly multi-arch image to DockerHub"
|
- label: "Publish nightly multi-arch image to DockerHub"
|
||||||
depends_on:
|
depends_on:
|
||||||
- create-multi-arch-manifest
|
- create-multi-arch-manifest
|
||||||
@@ -819,6 +848,23 @@ steps:
|
|||||||
DOCKER_BUILDKIT: "1"
|
DOCKER_BUILDKIT: "1"
|
||||||
S3_BUCKET: "vllm-wheels"
|
S3_BUCKET: "vllm-wheels"
|
||||||
|
|
||||||
|
- label: "Publish nightly XPU image to DockerHub"
|
||||||
|
depends_on:
|
||||||
|
- create-manifest-xpu
|
||||||
|
if: build.env("NIGHTLY") == "1"
|
||||||
|
agents:
|
||||||
|
queue: small_cpu_queue_release
|
||||||
|
commands:
|
||||||
|
- "bash .buildkite/scripts/xpu/push-nightly-builds-xpu.sh"
|
||||||
|
- "bash .buildkite/scripts/cleanup-nightly-builds.sh nightly- vllm/vllm-openai-xpu"
|
||||||
|
plugins:
|
||||||
|
- docker-login#v3.0.0:
|
||||||
|
username: vllmbot
|
||||||
|
password-env: DOCKERHUB_TOKEN
|
||||||
|
env:
|
||||||
|
DOCKER_BUILDKIT: "1"
|
||||||
|
DOCKERHUB_USERNAME: "vllmbot"
|
||||||
|
|
||||||
- label: "Publish nightly ROCm image to DockerHub"
|
- label: "Publish nightly ROCm image to DockerHub"
|
||||||
depends_on:
|
depends_on:
|
||||||
- build-rocm-release-image
|
- build-rocm-release-image
|
||||||
@@ -849,6 +895,7 @@ steps:
|
|||||||
- create-multi-arch-manifest-cuda-12-9
|
- create-multi-arch-manifest-cuda-12-9
|
||||||
- create-multi-arch-manifest-ubuntu2404
|
- create-multi-arch-manifest-ubuntu2404
|
||||||
- create-multi-arch-manifest-cuda-12-9-ubuntu2404
|
- create-multi-arch-manifest-cuda-12-9-ubuntu2404
|
||||||
|
- create-manifest-xpu
|
||||||
- build-rocm-release-image
|
- build-rocm-release-image
|
||||||
- input-release-version
|
- input-release-version
|
||||||
# Wait for CPU builds if their block steps were unblocked, so publish
|
# Wait for CPU builds if their block steps were unblocked, so publish
|
||||||
|
|||||||
Executable
+36
@@ -0,0 +1,36 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
#
|
||||||
|
# Append the Docker image tag(s) an image-build step pushed to a Buildkite
|
||||||
|
# annotation, so the built image tags show up on the build page instead of
|
||||||
|
# being buried in the job logs.
|
||||||
|
#
|
||||||
|
# Usage: annotate-image-build.sh <image_tag> [<image_tag> ...]
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# buildkite-agent only exists on Buildkite agents; no-op elsewhere so the
|
||||||
|
# image build scripts stay runnable locally.
|
||||||
|
if ! command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
|
echo "buildkite-agent not found; skipping image tag annotation"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
label="${BUILDKITE_LABEL:-Image build}"
|
||||||
|
content=""
|
||||||
|
for image in "$@"; do
|
||||||
|
[[ -n "$image" ]] || continue
|
||||||
|
content+="- **${label}**: \`${image}\`"$'\n'
|
||||||
|
done
|
||||||
|
|
||||||
|
if [[ -z "$content" ]]; then
|
||||||
|
echo "No image tags provided; nothing to annotate"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Best-effort: a flaky annotation must never fail an otherwise successful
|
||||||
|
# (and expensive) image build.
|
||||||
|
if ! printf '%s' "$content" | \
|
||||||
|
buildkite-agent annotate --append --style 'info' --context 'docker-images'; then
|
||||||
|
echo "warning: failed to annotate build with image tags"
|
||||||
|
fi
|
||||||
@@ -29,7 +29,11 @@ if python3 -c "import torch; assert torch.version.hip" 2>/dev/null; then
|
|||||||
TORCH_INDEX_URL=""
|
TORCH_INDEX_URL=""
|
||||||
fi
|
fi
|
||||||
else
|
else
|
||||||
TORCH_INDEX_URL="https://download.pytorch.org/whl/cu130"
|
if [ "${TORCH_NIGHTLY:-0}" = "1" ]; then
|
||||||
|
TORCH_INDEX_URL="https://download.pytorch.org/whl/nightly/cu130"
|
||||||
|
else
|
||||||
|
TORCH_INDEX_URL="https://download.pytorch.org/whl/cu130"
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
echo ">>> Using PyTorch index: ${TORCH_INDEX_URL:-PyPI default}"
|
echo ">>> Using PyTorch index: ${TORCH_INDEX_URL:-PyPI default}"
|
||||||
|
|
||||||
|
|||||||
@@ -15,9 +15,10 @@ set -euo pipefail
|
|||||||
|
|
||||||
DEFAULT_REPO_SLUG="vllm-project/vllm"
|
DEFAULT_REPO_SLUG="vllm-project/vllm"
|
||||||
DEFAULT_CI_HCL_SOURCE="docker/ci-rocm.hcl"
|
DEFAULT_CI_HCL_SOURCE="docker/ci-rocm.hcl"
|
||||||
DEFAULT_CI_BASE_CONTENT_FILES="requirements/common.txt requirements/rocm.txt requirements/test/rocm.txt docker/Dockerfile.rocm_base tools/install_torchcodec_rocm.sh tests/vllm_test_utils"
|
DEFAULT_CI_BASE_CONTENT_FILES="requirements/common.txt requirements/rocm.txt requirements/test/rocm.txt docker/Dockerfile.rocm_base docker/ci-rocm.hcl docker/docker-bake-rocm.hcl tools/install_torchcodec_rocm.sh tests/vllm_test_utils .buildkite/scripts/ci-bake-rocm.sh .buildkite/scripts/rocm/build-ci-base.sh"
|
||||||
DEFAULT_CI_BASE_DOCKERFILE="docker/Dockerfile.rocm"
|
DEFAULT_CI_BASE_DOCKERFILE="docker/Dockerfile.rocm"
|
||||||
DEFAULT_CI_BASE_DOCKERFILE_STAGES="base build_rixl build_rocshmem build_deepep mori_base ci_base"
|
DEFAULT_CI_BASE_DOCKERFILE_STAGES="base build_rixl build_rocshmem build_deepep mori_base ci_base"
|
||||||
|
DEFAULT_CI_BASE_METADATA_VERSION="1"
|
||||||
IMAGE_EXISTED_BEFORE_BUILD=0
|
IMAGE_EXISTED_BEFORE_BUILD=0
|
||||||
|
|
||||||
TARGET=""
|
TARGET=""
|
||||||
@@ -392,6 +393,16 @@ should_upload_wheel_artifacts() {
|
|||||||
|| "${TARGET}" == *"artifact"* ]]
|
|| "${TARGET}" == *"artifact"* ]]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
set_buildkite_metadata() {
|
||||||
|
local key="$1"
|
||||||
|
local value="$2"
|
||||||
|
|
||||||
|
[[ -n "${value}" ]] || return 0
|
||||||
|
if command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
|
buildkite-agent meta-data set "${key}" "${value}" || true
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
get_remote_image_label() {
|
get_remote_image_label() {
|
||||||
local image_ref="$1"
|
local image_ref="$1"
|
||||||
local label_key="$2"
|
local label_key="$2"
|
||||||
@@ -525,6 +536,22 @@ get_remote_image_label_with_retry() {
|
|||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
remote_ci_base_metadata_is_current() {
|
||||||
|
local image_ref="$1"
|
||||||
|
local metadata_version=""
|
||||||
|
|
||||||
|
metadata_version=$(get_remote_image_label "${image_ref}" "vllm.ci_base.metadata_version")
|
||||||
|
[[ "${metadata_version}" == "${CI_BASE_METADATA_VERSION:-${DEFAULT_CI_BASE_METADATA_VERSION}}" ]]
|
||||||
|
}
|
||||||
|
|
||||||
|
remote_ci_base_metadata_is_current_with_retry() {
|
||||||
|
local image_ref="$1"
|
||||||
|
local metadata_version=""
|
||||||
|
|
||||||
|
metadata_version=$(get_remote_image_label_with_retry "${image_ref}" "vllm.ci_base.metadata_version")
|
||||||
|
[[ "${metadata_version}" == "${CI_BASE_METADATA_VERSION:-${DEFAULT_CI_BASE_METADATA_VERSION}}" ]]
|
||||||
|
}
|
||||||
|
|
||||||
remote_image_exists() {
|
remote_image_exists() {
|
||||||
local image_ref="$1"
|
local image_ref="$1"
|
||||||
docker manifest inspect "${image_ref}" >/dev/null 2>&1
|
docker manifest inspect "${image_ref}" >/dev/null 2>&1
|
||||||
@@ -581,6 +608,7 @@ init_config() {
|
|||||||
CI_BASE_CONTENT_FILES="${CI_BASE_CONTENT_FILES:-${DEFAULT_CI_BASE_CONTENT_FILES}}"
|
CI_BASE_CONTENT_FILES="${CI_BASE_CONTENT_FILES:-${DEFAULT_CI_BASE_CONTENT_FILES}}"
|
||||||
CI_BASE_DOCKERFILE="${CI_BASE_DOCKERFILE:-${DEFAULT_CI_BASE_DOCKERFILE}}"
|
CI_BASE_DOCKERFILE="${CI_BASE_DOCKERFILE:-${DEFAULT_CI_BASE_DOCKERFILE}}"
|
||||||
CI_BASE_DOCKERFILE_STAGES="${CI_BASE_DOCKERFILE_STAGES:-${DEFAULT_CI_BASE_DOCKERFILE_STAGES}}"
|
CI_BASE_DOCKERFILE_STAGES="${CI_BASE_DOCKERFILE_STAGES:-${DEFAULT_CI_BASE_DOCKERFILE_STAGES}}"
|
||||||
|
CI_BASE_METADATA_VERSION="${CI_BASE_METADATA_VERSION:-${DEFAULT_CI_BASE_METADATA_VERSION}}"
|
||||||
CI_BASE_IMAGE_TAG="${CI_BASE_IMAGE_TAG:-rocm/vllm-dev:ci_base}"
|
CI_BASE_IMAGE_TAG="${CI_BASE_IMAGE_TAG:-rocm/vllm-dev:ci_base}"
|
||||||
export PYTORCH_ROCM_ARCH
|
export PYTORCH_ROCM_ARCH
|
||||||
|
|
||||||
@@ -635,6 +663,10 @@ load_ci_hcl() {
|
|||||||
echo "Copied ${CI_HCL_SOURCE} to ${CI_HCL_PATH}"
|
echo "Copied ${CI_HCL_SOURCE} to ${CI_HCL_PATH}"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
init_bake_files() {
|
||||||
|
BAKE_FILES=(-f "${VLLM_BAKE_FILE}" -f "${CI_HCL_PATH}")
|
||||||
|
}
|
||||||
|
|
||||||
compute_ci_base_hash_if_needed() {
|
compute_ci_base_hash_if_needed() {
|
||||||
if [[ -z "${CI_BASE_CONTENT_FILES:-}" ]]; then
|
if [[ -z "${CI_BASE_CONTENT_FILES:-}" ]]; then
|
||||||
return 0
|
return 0
|
||||||
@@ -676,12 +708,14 @@ configure_ci_base_image_refs() {
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
content_tag=$(ci_base_tag_with_suffix "${stable_tag}" "${CI_BASE_CONTENT_HASH}")
|
content_tag=$(ci_base_tag_with_suffix "${stable_tag}" "${CI_BASE_CONTENT_HASH}")
|
||||||
|
CI_BASE_IMAGE_TAG_CONTENT_REF="${content_tag}"
|
||||||
if [[ -n "${BUILDKITE_COMMIT:-}" ]]; then
|
if [[ -n "${BUILDKITE_COMMIT:-}" ]]; then
|
||||||
commit_tag=$(ci_base_tag_with_suffix "${stable_tag}" "${BUILDKITE_COMMIT}")
|
commit_tag=$(ci_base_tag_with_suffix "${stable_tag}" "${BUILDKITE_COMMIT}")
|
||||||
CI_BASE_IMAGE_TAG_COMMIT="${commit_tag}"
|
|
||||||
export CI_BASE_IMAGE_TAG_COMMIT
|
|
||||||
fi
|
fi
|
||||||
|
CI_BASE_IMAGE_TAG_COMMIT_REF="${commit_tag}"
|
||||||
|
|
||||||
|
# *_REF is the logical tag recorded in metadata. *_EXTRA is only passed to
|
||||||
|
# bake when that tag is not already the primary tag, avoiding duplicates.
|
||||||
if should_push_stable_ci_base_tag; then
|
if should_push_stable_ci_base_tag; then
|
||||||
primary_tag="${content_tag}"
|
primary_tag="${content_tag}"
|
||||||
CI_BASE_IMAGE_TAG_STABLE="${stable_tag}"
|
CI_BASE_IMAGE_TAG_STABLE="${stable_tag}"
|
||||||
@@ -691,19 +725,35 @@ configure_ci_base_image_refs() {
|
|||||||
fi
|
fi
|
||||||
CI_BASE_IMAGE_TAG="${primary_tag}"
|
CI_BASE_IMAGE_TAG="${primary_tag}"
|
||||||
if [[ "${primary_tag}" == "${content_tag}" ]]; then
|
if [[ "${primary_tag}" == "${content_tag}" ]]; then
|
||||||
CI_BASE_IMAGE_TAG_CONTENT=""
|
CI_BASE_IMAGE_TAG_CONTENT_EXTRA=""
|
||||||
else
|
else
|
||||||
CI_BASE_IMAGE_TAG_CONTENT="${content_tag}"
|
CI_BASE_IMAGE_TAG_CONTENT_EXTRA="${content_tag}"
|
||||||
fi
|
fi
|
||||||
export CI_BASE_IMAGE_TAG CI_BASE_IMAGE_TAG_CONTENT CI_BASE_IMAGE_TAG_STABLE
|
if [[ -n "${commit_tag}" && "${commit_tag}" != "${primary_tag}" ]]; then
|
||||||
|
CI_BASE_IMAGE_TAG_COMMIT_EXTRA="${commit_tag}"
|
||||||
|
else
|
||||||
|
CI_BASE_IMAGE_TAG_COMMIT_EXTRA=""
|
||||||
|
fi
|
||||||
|
export CI_BASE_IMAGE_TAG
|
||||||
|
export CI_BASE_IMAGE_TAG_COMMIT_EXTRA
|
||||||
|
export CI_BASE_IMAGE_TAG_CONTENT_EXTRA
|
||||||
|
export CI_BASE_IMAGE_TAG_CONTENT_REF
|
||||||
|
export CI_BASE_IMAGE_TAG_COMMIT_REF
|
||||||
|
export CI_BASE_IMAGE_TAG_STABLE
|
||||||
|
|
||||||
if is_ci_base_target; then
|
if is_ci_base_target; then
|
||||||
IMAGE_TAG="${primary_tag}"
|
IMAGE_TAG="${primary_tag}"
|
||||||
|
CI_BASE_IMAGE="${primary_tag}"
|
||||||
|
export CI_BASE_IMAGE
|
||||||
export IMAGE_TAG
|
export IMAGE_TAG
|
||||||
|
|
||||||
echo "ci_base primary image tag: ${CI_BASE_IMAGE_TAG}"
|
echo "ci_base primary image tag: ${CI_BASE_IMAGE_TAG}"
|
||||||
if [[ -n "${CI_BASE_IMAGE_TAG_COMMIT:-}" ]]; then
|
if [[ -n "${commit_tag}" ]]; then
|
||||||
echo "ci_base commit image tag: ${CI_BASE_IMAGE_TAG_COMMIT}"
|
if [[ "${commit_tag}" == "${primary_tag}" ]]; then
|
||||||
|
echo "ci_base commit image tag: ${commit_tag} (primary)"
|
||||||
|
else
|
||||||
|
echo "ci_base commit image tag: ${commit_tag}"
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
echo "ci_base content image tag: ${content_tag}"
|
echo "ci_base content image tag: ${content_tag}"
|
||||||
if [[ -n "${CI_BASE_IMAGE_TAG_STABLE}" ]]; then
|
if [[ -n "${CI_BASE_IMAGE_TAG_STABLE}" ]]; then
|
||||||
@@ -712,6 +762,10 @@ configure_ci_base_image_refs() {
|
|||||||
echo "ci_base stable alias will not be pushed for this build"
|
echo "ci_base stable alias will not be pushed for this build"
|
||||||
echo "Set NIGHTLY=1 on ${CI_BASE_STABLE_BRANCH:-main} to refresh ${stable_tag}"
|
echo "Set NIGHTLY=1 on ${CI_BASE_STABLE_BRANCH:-main} to refresh ${stable_tag}"
|
||||||
fi
|
fi
|
||||||
|
set_buildkite_metadata "rocm-ci-base-image" "${CI_BASE_IMAGE_TAG}"
|
||||||
|
set_buildkite_metadata "rocm-ci-base-image-content" "${content_tag}"
|
||||||
|
set_buildkite_metadata "rocm-ci-base-image-commit" "${CI_BASE_IMAGE_TAG_COMMIT:-}"
|
||||||
|
set_buildkite_metadata "rocm-ci-base-image-stable" "${CI_BASE_IMAGE_TAG_STABLE:-}"
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
@@ -728,8 +782,8 @@ ci_base_candidate_refs() {
|
|||||||
printf '%s\n' \
|
printf '%s\n' \
|
||||||
"${IMAGE_TAG:-}" \
|
"${IMAGE_TAG:-}" \
|
||||||
"${CI_BASE_IMAGE_TAG:-}" \
|
"${CI_BASE_IMAGE_TAG:-}" \
|
||||||
"${CI_BASE_IMAGE_TAG_COMMIT:-}" \
|
"${CI_BASE_IMAGE_TAG_COMMIT_EXTRA:-}" \
|
||||||
"${CI_BASE_IMAGE_TAG_CONTENT:-}" \
|
"${CI_BASE_IMAGE_TAG_CONTENT_EXTRA:-}" \
|
||||||
"${CI_BASE_IMAGE_TAG_STABLE:-}" \
|
"${CI_BASE_IMAGE_TAG_STABLE:-}" \
|
||||||
| awk 'NF && !seen[$0]++'
|
| awk 'NF && !seen[$0]++'
|
||||||
}
|
}
|
||||||
@@ -743,6 +797,10 @@ find_matching_ci_base_ref() {
|
|||||||
remote_image_exists "${candidate}" || continue
|
remote_image_exists "${candidate}" || continue
|
||||||
candidate_hash=$(get_remote_image_label "${candidate}" "vllm.ci_base.content_hash")
|
candidate_hash=$(get_remote_image_label "${candidate}" "vllm.ci_base.content_hash")
|
||||||
if [[ "${candidate_hash}" == "${CI_BASE_CONTENT_HASH}" ]]; then
|
if [[ "${candidate_hash}" == "${CI_BASE_CONTENT_HASH}" ]]; then
|
||||||
|
if ! remote_ci_base_metadata_is_current "${candidate}"; then
|
||||||
|
echo "Found matching ci_base content hash but stale metadata: ${candidate}" >&2
|
||||||
|
continue
|
||||||
|
fi
|
||||||
printf '%s\n' "${candidate}"
|
printf '%s\n' "${candidate}"
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
@@ -817,6 +875,10 @@ maybe_skip_existing_image() {
|
|||||||
if [[ -n "${remote_hash}" ]]; then
|
if [[ -n "${remote_hash}" ]]; then
|
||||||
echo "Remote ci_base content hash: ${remote_hash:0:16}..."
|
echo "Remote ci_base content hash: ${remote_hash:0:16}..."
|
||||||
if [[ "${remote_hash}" == "${CI_BASE_CONTENT_HASH}" ]]; then
|
if [[ "${remote_hash}" == "${CI_BASE_CONTENT_HASH}" ]]; then
|
||||||
|
if ! remote_ci_base_metadata_is_current "${IMAGE_TAG}"; then
|
||||||
|
echo "Content hashes match but ci_base metadata is stale; rebuilding to refresh metadata"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
if ! refresh_ci_base_tags_from_ref "${IMAGE_TAG}"; then
|
if ! refresh_ci_base_tags_from_ref "${IMAGE_TAG}"; then
|
||||||
echo "ci_base tag refresh failed; rebuilding to push expected tags"
|
echo "ci_base tag refresh failed; rebuilding to push expected tags"
|
||||||
return 0
|
return 0
|
||||||
@@ -998,12 +1060,104 @@ prepare_git_cache_metadata() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
ci_base_metadata_pairs() {
|
||||||
|
local dockerfile="${CI_BASE_DOCKERFILE:-${DEFAULT_CI_BASE_DOCKERFILE}}"
|
||||||
|
local stages="${CI_BASE_DOCKERFILE_STAGES:-${DEFAULT_CI_BASE_DOCKERFILE_STAGES}}"
|
||||||
|
local content_files="${CI_BASE_CONTENT_FILES:-${DEFAULT_CI_BASE_CONTENT_FILES}}"
|
||||||
|
local content_files_hash=""
|
||||||
|
local base_image=""
|
||||||
|
local base_image_digest=""
|
||||||
|
local git_branch=""
|
||||||
|
local -a content_paths=()
|
||||||
|
local -a content_args=()
|
||||||
|
|
||||||
|
read -r -a content_paths <<< "${content_files}"
|
||||||
|
if [[ ${#content_paths[@]} -gt 0 ]]; then
|
||||||
|
content_files_hash=$(compute_content_hash "${content_paths[@]}")
|
||||||
|
fi
|
||||||
|
mapfile -t content_args < <(
|
||||||
|
get_content_arg_names "${dockerfile}" "${stages}" "${CI_BASE_CONTENT_ARGS:-}"
|
||||||
|
)
|
||||||
|
|
||||||
|
base_image=$(resolve_dockerfile_arg_value "${dockerfile}" "BASE_IMAGE")
|
||||||
|
if [[ -n "${base_image}" ]]; then
|
||||||
|
base_image_digest=$(resolve_image_digest "${base_image}")
|
||||||
|
fi
|
||||||
|
git_branch="${BUILDKITE_BRANCH:-${VLLM_BRANCH:-}}"
|
||||||
|
|
||||||
|
metadata_pair "vllm.ci_base.metadata_version" "${CI_BASE_METADATA_VERSION:-${DEFAULT_CI_BASE_METADATA_VERSION}}"
|
||||||
|
metadata_pair "vllm.ci_base.content_hash" "${CI_BASE_CONTENT_HASH:-}"
|
||||||
|
metadata_pair "vllm.ci_base.content_files_hash" "${content_files_hash}"
|
||||||
|
metadata_pair "vllm.ci_base.content_files" "${content_files}"
|
||||||
|
metadata_pair "vllm.ci_base.content_args" "$(join_words "${content_args[@]}")"
|
||||||
|
metadata_pair "vllm.ci_base.dockerfile" "${dockerfile}"
|
||||||
|
metadata_pair "vllm.ci_base.dockerfile_stages" "${stages}"
|
||||||
|
metadata_pair "vllm.ci_base.image.primary" "${CI_BASE_IMAGE_TAG:-}"
|
||||||
|
metadata_pair "vllm.ci_base.image.content" "${CI_BASE_IMAGE_TAG_CONTENT_REF:-${CI_BASE_IMAGE_TAG_CONTENT_EXTRA:-}}"
|
||||||
|
metadata_pair "vllm.ci_base.image.commit" "${CI_BASE_IMAGE_TAG_COMMIT_REF:-${CI_BASE_IMAGE_TAG_COMMIT_EXTRA:-}}"
|
||||||
|
metadata_pair "vllm.ci_base.image.stable" "${CI_BASE_IMAGE_TAG_STABLE:-}"
|
||||||
|
metadata_pair "vllm.ci_base.git_commit" "${BUILDKITE_COMMIT:-}"
|
||||||
|
metadata_pair "vllm.ci_base.git_branch" "${git_branch}"
|
||||||
|
metadata_pair "vllm.ci_base.vllm_branch" "${VLLM_BRANCH:-}"
|
||||||
|
metadata_pair "vllm.ci_base.stable_branch" "${CI_BASE_STABLE_BRANCH:-main}"
|
||||||
|
|
||||||
|
metadata_pair "vllm.rocm.base_image" "${base_image}"
|
||||||
|
metadata_pair "vllm.rocm.base_image_digest" "${base_image_digest}"
|
||||||
|
metadata_pair "vllm.rocm.pytorch_rocm_arch" "${PYTORCH_ROCM_ARCH:-}"
|
||||||
|
metadata_pair "vllm.rocm.nic_backend" "$(resolve_dockerfile_arg_value "${dockerfile}" "NIC_BACKEND")"
|
||||||
|
metadata_pair "vllm.rocm.ainic_version" "$(resolve_dockerfile_arg_value "${dockerfile}" "AINIC_VERSION")"
|
||||||
|
metadata_pair "vllm.rocm.ubuntu_codename" "$(resolve_dockerfile_arg_value "${dockerfile}" "UBUNTU_CODENAME")"
|
||||||
|
metadata_pair "vllm.rocm.rixl_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "RIXL_REPO")"
|
||||||
|
metadata_pair "vllm.rocm.rixl_commit" "${RIXL_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "RIXL_BRANCH")}"
|
||||||
|
metadata_pair "vllm.rocm.ucx_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "UCX_REPO")"
|
||||||
|
metadata_pair "vllm.rocm.ucx_commit" "${UCX_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "UCX_BRANCH")}"
|
||||||
|
metadata_pair "vllm.rocm.rocshmem_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "ROCSHMEM_REPO")"
|
||||||
|
metadata_pair "vllm.rocm.rocshmem_commit" "${ROCSHMEM_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "ROCSHMEM_BRANCH")}"
|
||||||
|
metadata_pair "vllm.rocm.deepep_repo" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_REPO")"
|
||||||
|
metadata_pair "vllm.rocm.deepep_commit" "${DEEPEP_BRANCH:-$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_BRANCH")}"
|
||||||
|
metadata_pair "vllm.rocm.deepep_nic" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_NIC")"
|
||||||
|
metadata_pair "vllm.rocm.deepep_rocm_arch" "$(resolve_dockerfile_arg_value "${dockerfile}" "DEEPEP_ROCM_ARCH")"
|
||||||
|
metadata_pair "vllm.rocm.rixl_cache_key" "${RIXL_CACHE_KEY:-}"
|
||||||
|
metadata_pair "vllm.rocm.rocshmem_cache_key" "${ROCSHMEM_CACHE_KEY:-}"
|
||||||
|
metadata_pair "vllm.rocm.deepep_cache_key" "${DEEPEP_CACHE_KEY:-}"
|
||||||
|
|
||||||
|
metadata_pair "vllm.buildkite.build_number" "${BUILDKITE_BUILD_NUMBER:-}"
|
||||||
|
metadata_pair "vllm.buildkite.build_id" "${BUILDKITE_BUILD_ID:-}"
|
||||||
|
}
|
||||||
|
|
||||||
|
write_ci_base_metadata_annotations() {
|
||||||
|
local metadata="$1"
|
||||||
|
local key=""
|
||||||
|
local value=""
|
||||||
|
local annotation=""
|
||||||
|
|
||||||
|
[[ -n "${metadata}" ]] || return 0
|
||||||
|
while IFS=$'\t' read -r key value; do
|
||||||
|
[[ -n "${key}" && -n "${value}" ]] || continue
|
||||||
|
annotation="manifest:${key}=${value}"
|
||||||
|
printf ' "%s",\n' "$(hcl_escape_string "${annotation}")"
|
||||||
|
done <<< "${metadata}"
|
||||||
|
}
|
||||||
|
|
||||||
|
write_ci_base_metadata_labels() {
|
||||||
|
local metadata="$1"
|
||||||
|
local key=""
|
||||||
|
local value=""
|
||||||
|
|
||||||
|
[[ -n "${metadata}" ]] || return 0
|
||||||
|
while IFS=$'\t' read -r key value; do
|
||||||
|
[[ -n "${key}" && -n "${value}" ]] || continue
|
||||||
|
printf ' "%s" = "%s"\n' \
|
||||||
|
"$(hcl_escape_string "${key}")" \
|
||||||
|
"$(hcl_escape_string "${value}")"
|
||||||
|
done <<< "${metadata}"
|
||||||
|
}
|
||||||
|
|
||||||
write_ci_base_label_override() {
|
write_ci_base_label_override() {
|
||||||
local target_name=""
|
local target_name=""
|
||||||
|
local metadata=""
|
||||||
local -a ci_base_targets=()
|
local -a ci_base_targets=()
|
||||||
|
|
||||||
BAKE_FILES=(-f "${VLLM_BAKE_FILE}" -f "${CI_HCL_PATH}")
|
|
||||||
|
|
||||||
if [[ -z "${CI_BASE_CONTENT_HASH:-}" ]]; then
|
if [[ -z "${CI_BASE_CONTENT_HASH:-}" ]]; then
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
@@ -1019,16 +1173,23 @@ write_ci_base_label_override() {
|
|||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
metadata=$(ci_base_metadata_pairs)
|
||||||
|
|
||||||
: > "${CI_BASE_LABEL_OVERRIDE_PATH}"
|
: > "${CI_BASE_LABEL_OVERRIDE_PATH}"
|
||||||
for target_name in "${ci_base_targets[@]}"; do
|
for target_name in "${ci_base_targets[@]}"; do
|
||||||
cat >> "${CI_BASE_LABEL_OVERRIDE_PATH}" <<EOF
|
cat >> "${CI_BASE_LABEL_OVERRIDE_PATH}" <<EOF
|
||||||
target "${target_name}" {
|
target "${target_name}" {
|
||||||
annotations = [
|
annotations = [
|
||||||
"manifest:org.opencontainers.image.revision=",
|
"manifest:org.opencontainers.image.revision=",
|
||||||
|
EOF
|
||||||
|
write_ci_base_metadata_annotations "${metadata}" >> "${CI_BASE_LABEL_OVERRIDE_PATH}"
|
||||||
|
cat >> "${CI_BASE_LABEL_OVERRIDE_PATH}" <<EOF
|
||||||
]
|
]
|
||||||
labels = {
|
labels = {
|
||||||
"org.opencontainers.image.revision" = ""
|
"org.opencontainers.image.revision" = ""
|
||||||
"vllm.ci_base.content_hash" = "${CI_BASE_CONTENT_HASH}"
|
EOF
|
||||||
|
write_ci_base_metadata_labels "${metadata}" >> "${CI_BASE_LABEL_OVERRIDE_PATH}"
|
||||||
|
cat >> "${CI_BASE_LABEL_OVERRIDE_PATH}" <<EOF
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1036,7 +1197,7 @@ EOF
|
|||||||
done
|
done
|
||||||
|
|
||||||
BAKE_FILES+=(-f "${CI_BASE_LABEL_OVERRIDE_PATH}")
|
BAKE_FILES+=(-f "${CI_BASE_LABEL_OVERRIDE_PATH}")
|
||||||
echo "Appended ci_base content-hash label override for targets: ${ci_base_targets[*]}"
|
echo "Appended ci_base metadata label override for targets: ${ci_base_targets[*]}"
|
||||||
}
|
}
|
||||||
|
|
||||||
uses_rocm_csrc_cache() {
|
uses_rocm_csrc_cache() {
|
||||||
@@ -1119,6 +1280,18 @@ hcl_escape_string() {
|
|||||||
printf '%s' "${value}"
|
printf '%s' "${value}"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
join_words() {
|
||||||
|
local IFS=" "
|
||||||
|
printf '%s' "$*"
|
||||||
|
}
|
||||||
|
|
||||||
|
metadata_pair() {
|
||||||
|
local key="$1"
|
||||||
|
local value="${2:-}"
|
||||||
|
|
||||||
|
printf '%s\t%s\n' "${key}" "${value}"
|
||||||
|
}
|
||||||
|
|
||||||
write_hcl_string_list() {
|
write_hcl_string_list() {
|
||||||
local indent="$1"
|
local indent="$1"
|
||||||
shift
|
shift
|
||||||
@@ -1541,7 +1714,13 @@ confirm_remote_image_push() {
|
|||||||
|
|
||||||
remote_hash=$(get_remote_image_label_with_retry "${image_ref}" "vllm.ci_base.content_hash")
|
remote_hash=$(get_remote_image_label_with_retry "${image_ref}" "vllm.ci_base.content_hash")
|
||||||
if [[ -n "${remote_hash}" && "${remote_hash}" == "${CI_BASE_CONTENT_HASH}" ]]; then
|
if [[ -n "${remote_hash}" && "${remote_hash}" == "${CI_BASE_CONTENT_HASH}" ]]; then
|
||||||
return 0
|
if remote_ci_base_metadata_is_current_with_retry "${image_ref}"; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Remote image exists with the expected ci_base content hash but stale metadata."
|
||||||
|
echo " expected metadata version: ${CI_BASE_METADATA_VERSION:-${DEFAULT_CI_BASE_METADATA_VERSION}}"
|
||||||
|
return 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo "Remote image exists but does not have the expected ci_base content hash."
|
echo "Remote image exists but does not have the expected ci_base content hash."
|
||||||
@@ -1616,7 +1795,10 @@ seed_dependency_caches_if_needed() {
|
|||||||
|
|
||||||
echo "--- :docker: Seeding ${target}"
|
echo "--- :docker: Seeding ${target}"
|
||||||
echo "Expected cache ref: ${cache_ref}"
|
echo "Expected cache ref: ${cache_ref}"
|
||||||
docker buildx bake "${BAKE_FILES[@]}" --progress plain "${target}"
|
docker buildx bake \
|
||||||
|
"${BAKE_FILES[@]}" \
|
||||||
|
--progress "${BUILDKIT_PROGRESS:-plain}" \
|
||||||
|
"${target}"
|
||||||
verify_dependency_cache_ref "${cache_ref}"
|
verify_dependency_cache_ref "${cache_ref}"
|
||||||
done
|
done
|
||||||
}
|
}
|
||||||
@@ -1644,7 +1826,10 @@ run_bake() {
|
|||||||
local build_rc=0
|
local build_rc=0
|
||||||
|
|
||||||
echo "--- :docker: Building ${TARGET}"
|
echo "--- :docker: Building ${TARGET}"
|
||||||
docker buildx bake "${BAKE_FILES[@]}" --progress plain "${BAKE_TARGETS[@]}" || build_rc=$?
|
docker buildx bake \
|
||||||
|
"${BAKE_FILES[@]}" \
|
||||||
|
--progress "${BUILDKIT_PROGRESS:-plain}" \
|
||||||
|
"${BAKE_TARGETS[@]}" || build_rc=$?
|
||||||
|
|
||||||
if [[ ${build_rc} -eq 0 ]]; then
|
if [[ ${build_rc} -eq 0 ]]; then
|
||||||
echo "--- :white_check_mark: Build complete"
|
echo "--- :white_check_mark: Build complete"
|
||||||
@@ -1724,15 +1909,16 @@ main() {
|
|||||||
print_header
|
print_header
|
||||||
validate_inputs
|
validate_inputs
|
||||||
load_ci_hcl
|
load_ci_hcl
|
||||||
|
init_bake_files
|
||||||
compute_ci_base_hash_if_needed
|
compute_ci_base_hash_if_needed
|
||||||
configure_ci_base_image_refs
|
configure_ci_base_image_refs
|
||||||
maybe_skip_existing_image
|
maybe_skip_existing_image
|
||||||
setup_builder
|
setup_builder
|
||||||
prepare_git_cache_metadata
|
prepare_git_cache_metadata
|
||||||
write_ci_base_label_override
|
|
||||||
extract_dependency_pins
|
extract_dependency_pins
|
||||||
write_rocm_build_arg_override
|
write_rocm_build_arg_override
|
||||||
compute_dependency_cache_keys
|
compute_dependency_cache_keys
|
||||||
|
write_ci_base_label_override
|
||||||
compute_rocm_csrc_content_hash_if_needed
|
compute_rocm_csrc_content_hash_if_needed
|
||||||
write_rocm_cache_override
|
write_rocm_cache_override
|
||||||
resolve_ci_base_dependency_targets
|
resolve_ci_base_dependency_targets
|
||||||
|
|||||||
@@ -28,6 +28,17 @@
|
|||||||
###############################################################################
|
###############################################################################
|
||||||
set -o pipefail
|
set -o pipefail
|
||||||
|
|
||||||
|
: "${BUILDKIT_PROGRESS:=plain}"
|
||||||
|
: "${TERM:=xterm-256color}"
|
||||||
|
: "${FORCE_COLOR:=1}"
|
||||||
|
: "${CLICOLOR_FORCE:=1}"
|
||||||
|
: "${PY_COLORS:=1}"
|
||||||
|
: "${ROCM_DOCKER_TTY:=1}"
|
||||||
|
if [[ " ${PYTEST_ADDOPTS:-} " != *" --color"* ]]; then
|
||||||
|
PYTEST_ADDOPTS="${PYTEST_ADDOPTS:+${PYTEST_ADDOPTS} }--color=yes"
|
||||||
|
fi
|
||||||
|
export BUILDKIT_PROGRESS TERM FORCE_COLOR CLICOLOR_FORCE PY_COLORS PYTEST_ADDOPTS ROCM_DOCKER_TTY
|
||||||
|
|
||||||
# Export Python path for commands that run directly on the host. Containerized
|
# Export Python path for commands that run directly on the host. Containerized
|
||||||
# tests set this to /vllm-workspace below so spawned Python processes do not
|
# tests set this to /vllm-workspace below so spawned Python processes do not
|
||||||
# depend on their current working directory.
|
# depend on their current working directory.
|
||||||
@@ -149,6 +160,7 @@ EOF
|
|||||||
echo "--- Building local ROCm test image"
|
echo "--- Building local ROCm test image"
|
||||||
docker build \
|
docker build \
|
||||||
--pull=false \
|
--pull=false \
|
||||||
|
--progress "${BUILDKIT_PROGRESS}" \
|
||||||
--build-arg "BASE_IMAGE=${base_image}" \
|
--build-arg "BASE_IMAGE=${base_image}" \
|
||||||
-t "${artifact_image}" \
|
-t "${artifact_image}" \
|
||||||
"${context_dir}" || return 1
|
"${context_dir}" || return 1
|
||||||
@@ -367,6 +379,20 @@ remove_docker_container() {
|
|||||||
}
|
}
|
||||||
trap remove_docker_container EXIT
|
trap remove_docker_container EXIT
|
||||||
|
|
||||||
|
# python_only_compile.sh runs `python setup.py develop` and needs the full repo tree
|
||||||
|
# under /vllm-workspace (Dockerfile.rocm test stage: mkdir src && mv vllm).
|
||||||
|
# The ROCm wheel artifact tarball only ships a thin tree (tests, etc.), so
|
||||||
|
# artifact images cannot satisfy that test — use the full rocm/vllm-ci image.
|
||||||
|
_cmd_probe="${VLLM_TEST_COMMANDS:-}"
|
||||||
|
if [[ -z "${_cmd_probe}" ]]; then
|
||||||
|
_cmd_probe="$*"
|
||||||
|
fi
|
||||||
|
if [[ "${VLLM_CI_USE_ARTIFACTS:-0}" == "1" && "${_cmd_probe}" == *python_only_compile.sh* ]]; then
|
||||||
|
echo "INFO: disabling VLLM_CI_USE_ARTIFACTS for python_only_compile (requires full /vllm-workspace tree)"
|
||||||
|
export VLLM_CI_USE_ARTIFACTS=0
|
||||||
|
fi
|
||||||
|
unset -v _cmd_probe
|
||||||
|
|
||||||
if ! prepare_artifact_image; then
|
if ! prepare_artifact_image; then
|
||||||
echo "Using full ROCm CI image: ${image_name}"
|
echo "Using full ROCm CI image: ${image_name}"
|
||||||
docker pull "${image_name}" || exit 1
|
docker pull "${image_name}" || exit 1
|
||||||
@@ -426,6 +452,26 @@ fi
|
|||||||
|
|
||||||
echo "Final commands: $commands"
|
echo "Final commands: $commands"
|
||||||
|
|
||||||
|
# The ROCm test image often ships /vllm-workspace without .git (artifact tarball unpack).
|
||||||
|
# tests/standalone_tests/python_only_compile.sh uses merge-base(HEAD, origin/main) for
|
||||||
|
# wheels.vllm.ai; compute on the agent (full git checkout) and pass into the container.
|
||||||
|
vllm_standalone_merge_base=""
|
||||||
|
checkout="${BUILDKITE_BUILD_CHECKOUT_PATH:-}"
|
||||||
|
if [[ -z "${checkout}" || ! -d "${checkout}" ]]; then
|
||||||
|
checkout="."
|
||||||
|
fi
|
||||||
|
# Pass safe.directory per-command (-c) because buildkite runs will always fail
|
||||||
|
# the next check on git 2.35.2+ due to mixed uses of root and buildkite-agent/uids.
|
||||||
|
if git -c "safe.directory=${checkout}" -C "${checkout}" rev-parse --is-inside-work-tree >/dev/null 2>&1; then
|
||||||
|
vllm_standalone_merge_base="$(
|
||||||
|
git -c "safe.directory=${checkout}" -C "${checkout}" merge-base HEAD origin/main 2>/dev/null || true
|
||||||
|
)"
|
||||||
|
fi
|
||||||
|
if [[ -z "${vllm_standalone_merge_base}" ]]; then
|
||||||
|
vllm_standalone_merge_base="${BUILDKITE_COMMIT:-}"
|
||||||
|
fi
|
||||||
|
echo "INFO: passing VLLM_STANDALONE_MERGE_BASE into container: ${vllm_standalone_merge_base}"
|
||||||
|
|
||||||
MYPYTHONPATH="/vllm-workspace"
|
MYPYTHONPATH="/vllm-workspace"
|
||||||
|
|
||||||
container_job_id="${BUILDKITE_JOB_ID:-${BUILDKITE_PARALLEL_JOB:-0}}"
|
container_job_id="${BUILDKITE_JOB_ID:-${BUILDKITE_PARALLEL_JOB:-0}}"
|
||||||
@@ -501,14 +547,37 @@ if is_multi_node "$commands"; then
|
|||||||
else
|
else
|
||||||
echo "--- Single-node job"
|
echo "--- Single-node job"
|
||||||
echo "Render devices: $BUILDKITE_AGENT_META_DATA_RENDER_DEVICES"
|
echo "Render devices: $BUILDKITE_AGENT_META_DATA_RENDER_DEVICES"
|
||||||
|
docker_run_terminal_args=(-i)
|
||||||
|
if [[ "${ROCM_DOCKER_TTY}" == "1" ]]; then
|
||||||
|
docker_run_terminal_args+=(-t)
|
||||||
|
echo "Docker interactive stdin: enabled; TTY allocation: enabled"
|
||||||
|
else
|
||||||
|
echo "Docker interactive stdin: enabled; TTY allocation: disabled"
|
||||||
|
fi
|
||||||
|
|
||||||
|
ulimit_core_hard=$(ulimit -H -c)
|
||||||
|
if [[ "$ulimit_core_hard" == "unlimited" ]]; then
|
||||||
|
# docker run can't pass "unlimited" to --ulimit
|
||||||
|
ulimit_core_hard="-1"
|
||||||
|
fi
|
||||||
|
# Disable core dumps in the ROCm test container unless the ROCm debug agent is enabled
|
||||||
|
coredump_flags="--ulimit core=0:$ulimit_core_hard"
|
||||||
|
if [[ "$commands" == *"ROCm debug agent enabled"* ]]; then
|
||||||
|
# Works around https://github.com/rocm/rocm-systems/issues/6206
|
||||||
|
coredump_flags='-e HSA_COREDUMP_PATTERN="/tmp/gpucore.%p"'
|
||||||
|
else
|
||||||
|
echo "ROCm debug agent not enabled, coredumps are disabled in the test container."
|
||||||
|
fi
|
||||||
|
|
||||||
docker run \
|
docker run \
|
||||||
|
"${docker_run_terminal_args[@]}" \
|
||||||
--device /dev/kfd $BUILDKITE_AGENT_META_DATA_RENDER_DEVICES \
|
--device /dev/kfd $BUILDKITE_AGENT_META_DATA_RENDER_DEVICES \
|
||||||
$RDMA_FLAGS \
|
$RDMA_FLAGS \
|
||||||
--network=host \
|
--network=host \
|
||||||
--shm-size=16gb \
|
--shm-size=16gb \
|
||||||
--group-add "$render_gid" \
|
--group-add "$render_gid" \
|
||||||
--rm \
|
--rm \
|
||||||
|
$coredump_flags \
|
||||||
-e HF_TOKEN \
|
-e HF_TOKEN \
|
||||||
-e "HF_HUB_DOWNLOAD_TIMEOUT=${HF_HUB_DOWNLOAD_TIMEOUT}" \
|
-e "HF_HUB_DOWNLOAD_TIMEOUT=${HF_HUB_DOWNLOAD_TIMEOUT}" \
|
||||||
-e "HF_HUB_ETAG_TIMEOUT=${HF_HUB_ETAG_TIMEOUT}" \
|
-e "HF_HUB_ETAG_TIMEOUT=${HF_HUB_ETAG_TIMEOUT}" \
|
||||||
@@ -516,6 +585,11 @@ else
|
|||||||
-e AWS_SECRET_ACCESS_KEY \
|
-e AWS_SECRET_ACCESS_KEY \
|
||||||
-e BUILDKITE_PARALLEL_JOB \
|
-e BUILDKITE_PARALLEL_JOB \
|
||||||
-e BUILDKITE_PARALLEL_JOB_COUNT \
|
-e BUILDKITE_PARALLEL_JOB_COUNT \
|
||||||
|
-e TERM \
|
||||||
|
-e FORCE_COLOR \
|
||||||
|
-e CLICOLOR_FORCE \
|
||||||
|
-e PY_COLORS \
|
||||||
|
-e PYTEST_ADDOPTS \
|
||||||
-v "${HF_CACHE}:${HF_MOUNT}" \
|
-v "${HF_CACHE}:${HF_MOUNT}" \
|
||||||
-e "HF_HOME=${HF_MOUNT}" \
|
-e "HF_HOME=${HF_MOUNT}" \
|
||||||
-e "PYTHONPATH=${MYPYTHONPATH}" \
|
-e "PYTHONPATH=${MYPYTHONPATH}" \
|
||||||
@@ -525,6 +599,7 @@ else
|
|||||||
-e "VLLM_CACHE_ROOT=${CONTAINER_CACHE_ROOT}/vllm" \
|
-e "VLLM_CACHE_ROOT=${CONTAINER_CACHE_ROOT}/vllm" \
|
||||||
-e "XDG_CACHE_HOME=${CONTAINER_CACHE_ROOT}/xdg" \
|
-e "XDG_CACHE_HOME=${CONTAINER_CACHE_ROOT}/xdg" \
|
||||||
-e "PYTORCH_ROCM_ARCH=" \
|
-e "PYTORCH_ROCM_ARCH=" \
|
||||||
|
-e "VLLM_STANDALONE_MERGE_BASE=${vllm_standalone_merge_base}" \
|
||||||
--name "${container_name}" \
|
--name "${container_name}" \
|
||||||
"${image_name}" \
|
"${image_name}" \
|
||||||
/bin/bash -c "${CONTAINER_PREFLIGHT} && ${commands}"
|
/bin/bash -c "${CONTAINER_PREFLIGHT} && ${commands}"
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ set -ex
|
|||||||
CORE_RANGE=${CORE_RANGE:-0-31}
|
CORE_RANGE=${CORE_RANGE:-0-31}
|
||||||
OMP_CORE_RANGE=${OMP_CORE_RANGE:-0-31}
|
OMP_CORE_RANGE=${OMP_CORE_RANGE:-0-31}
|
||||||
|
|
||||||
export CMAKE_BUILD_PARALLEL_LEVEL=16
|
export CMAKE_BUILD_PARALLEL_LEVEL=32
|
||||||
|
|
||||||
# Setup cleanup
|
# Setup cleanup
|
||||||
remove_docker_container() {
|
remove_docker_container() {
|
||||||
@@ -37,8 +37,10 @@ function cpu_tests() {
|
|||||||
pytest -x -v -s tests/kernels/test_onednn.py
|
pytest -x -v -s tests/kernels/test_onednn.py
|
||||||
pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
|
pytest -x -v -s tests/kernels/attention/test_cpu_attn.py
|
||||||
pytest -x -v -s tests/kernels/core/test_cpu_activation.py
|
pytest -x -v -s tests/kernels/core/test_cpu_activation.py
|
||||||
pytest -x -v -s tests/kernels/moe/test_moe.py -k test_cpu_fused_moe_basic
|
pytest -x -v -s tests/kernels/moe/test_cpu_fused_moe.py
|
||||||
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py"
|
pytest -x -v -s tests/kernels/mamba/cpu/test_cpu_gdn_ops.py
|
||||||
|
pytest -x -v -s tests/kernels/moe/test_cpu_int4_moe.py
|
||||||
|
pytest -x -v -s tests/kernels/mamba/test_cpu_short_conv.py"
|
||||||
|
|
||||||
# skip tests requiring model downloads if HF_TOKEN is not set
|
# skip tests requiring model downloads if HF_TOKEN is not set
|
||||||
# due to rate-limits
|
# due to rate-limits
|
||||||
@@ -62,7 +64,6 @@ function cpu_tests() {
|
|||||||
set -e
|
set -e
|
||||||
pytest -x -v -s tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_logprobs"
|
pytest -x -v -s tests/quantization/test_compressed_tensors.py::test_compressed_tensors_w8a8_logprobs"
|
||||||
|
|
||||||
|
|
||||||
# basic online serving
|
# basic online serving
|
||||||
docker exec cpu-test bash -c '
|
docker exec cpu-test bash -c '
|
||||||
set -e
|
set -e
|
||||||
|
|||||||
@@ -7,10 +7,49 @@ set -euox pipefail
|
|||||||
# allow to bind to different cores
|
# allow to bind to different cores
|
||||||
CORE_RANGE=${CORE_RANGE:-48-95}
|
CORE_RANGE=${CORE_RANGE:-48-95}
|
||||||
NUMA_NODE=${NUMA_NODE:-1}
|
NUMA_NODE=${NUMA_NODE:-1}
|
||||||
IMAGE_NAME="cpu-test-$NUMA_NODE"
|
AGENT_SLOT=${AGENT_SLOT:-}
|
||||||
|
IMAGE_NAME="cpu-test-${NUMA_NODE}${AGENT_SLOT:+-${AGENT_SLOT}}"
|
||||||
TIMEOUT_VAL=$1
|
TIMEOUT_VAL=$1
|
||||||
TEST_COMMAND=$2
|
TEST_COMMAND=$2
|
||||||
|
|
||||||
|
# Disk hygiene knobs. Reclaim space only once the Docker root filesystem crosses
|
||||||
|
# DISK_USAGE_THRESHOLD percent, and cap the shared BuildKit cache at
|
||||||
|
# BUILDKIT_CACHE_MAX so subsequent builds keep reusing the hottest layers.
|
||||||
|
DISK_USAGE_THRESHOLD=${DISK_USAGE_THRESHOLD:-70}
|
||||||
|
BUILDKIT_CACHE_MAX=${BUILDKIT_CACHE_MAX:-80GB}
|
||||||
|
|
||||||
|
# Reclaim disk only when the host is under pressure. We trim (not purge) the
|
||||||
|
# shared BuildKit cache so cross-job/cross-agent reuse stays intact, and only
|
||||||
|
# touch dangling images; other agents' uniquely tagged images are left alone.
|
||||||
|
prune_if_disk_pressure() {
|
||||||
|
local docker_root disk_usage
|
||||||
|
docker_root=$(docker info -f '{{.DockerRootDir}}' 2>/dev/null || true)
|
||||||
|
if [ -z "$docker_root" ]; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
disk_usage=$(df "$docker_root" 2>/dev/null | tail -1 | awk '{print $5}' | tr -d '%')
|
||||||
|
if [ "${disk_usage:-0}" -gt "$DISK_USAGE_THRESHOLD" ]; then
|
||||||
|
echo "--- :broom: Disk usage ${disk_usage}% exceeds ${DISK_USAGE_THRESHOLD}%, reclaiming space"
|
||||||
|
docker image prune -f || true
|
||||||
|
docker builder prune -f --keep-storage="$BUILDKIT_CACHE_MAX" || true
|
||||||
|
else
|
||||||
|
echo "Disk usage ${disk_usage:-unknown}% within ${DISK_USAGE_THRESHOLD}% threshold; skipping prune"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
# Always drop this agent's image once the job ends (the default builder never
|
||||||
|
# uses it as a cache source, so removing it costs no rebuild speed), then
|
||||||
|
# reclaim space if needed. Guard every docker call with `|| true` so the trap
|
||||||
|
# never overrides the test's exit code.
|
||||||
|
cleanup() {
|
||||||
|
docker image rm -f "$IMAGE_NAME" || true
|
||||||
|
prune_if_disk_pressure
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
|
||||||
|
# Free space up front so a nearly-full host doesn't fail the build.
|
||||||
|
prune_if_disk_pressure
|
||||||
|
|
||||||
# building the docker image
|
# building the docker image
|
||||||
echo "--- :docker: Building Docker image"
|
echo "--- :docker: Building Docker image"
|
||||||
docker build --progress plain --tag "$IMAGE_NAME" --target vllm-test -f docker/Dockerfile.cpu .
|
docker build --progress plain --tag "$IMAGE_NAME" --target vllm-test -f docker/Dockerfile.cpu .
|
||||||
|
|||||||
@@ -0,0 +1,52 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
test_suite="${1:-}"
|
||||||
|
|
||||||
|
if [[ -z "${test_suite}" ]]; then
|
||||||
|
echo "Usage: $0 <example|v1|server>" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
case "${test_suite}" in
|
||||||
|
example)
|
||||||
|
pip install tblib==3.1.0
|
||||||
|
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 -O3 -cc.cudagraph_mode=NONE
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager -tp 2 --distributed-executor-backend mp
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager --attention-backend=TRITON_ATTN
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager --quantization fp8
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model facebook/opt-125m --block-size 64 --enforce-eager --kv-cache-dtype fp8
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model nvidia/Llama-3.1-8B-Instruct-FP8 --block-size 64 --enforce-eager --quantization modelopt --kv-cache-dtype fp8 --attention-backend TRITON_ATTN --max-model-len 4096
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model superjob/Qwen3-4B-Instruct-2507-GPTQ-Int4 --block-size 64 --enforce-eager --max-model-len 8192
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model TheBloke/TinyLlama-1.1B-Chat-v0.3-AWQ --block-size 64 --enforce-eager
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model ibm-research/PowerMoE-3b --block-size 64 --enforce-eager -tp 2
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model ibm-research/PowerMoE-3b --block-size 64 --enforce-eager -tp 2 --enable-expert-parallel
|
||||||
|
python3 examples/basic/offline_inference/generate.py --model superjob/Qwen3-4B-Instruct-2507-GPTQ-Int4 --max-model-len 8192
|
||||||
|
;;
|
||||||
|
v1)
|
||||||
|
cd tests
|
||||||
|
|
||||||
|
pytest -v -s v1/core --ignore=v1/core/test_reset_prefix_cache_e2e.py --ignore=v1/core/test_scheduler_e2e.py
|
||||||
|
pytest -v -s v1/engine --ignore=v1/engine/test_output_processor.py
|
||||||
|
pytest -v -s v1/sample --ignore=v1/sample/test_logprobs.py --ignore=v1/sample/test_logprobs_e2e.py -k "not test_topk_only and not test_topp_only and not test_topk_and_topp"
|
||||||
|
pytest -v -s v1/worker --ignore=v1/worker/test_gpu_model_runner.py --ignore=v1/worker/test_worker_memory_snapshot.py
|
||||||
|
pytest -v -s v1/structured_output
|
||||||
|
pytest -v -s v1/test_serial_utils.py
|
||||||
|
pytest -v -s v1/spec_decode --ignore=v1/spec_decode/test_max_len.py --ignore=v1/spec_decode/test_speculators_eagle3.py --ignore=v1/spec_decode/test_acceptance_length.py --ignore=v1/spec_decode/test_speculators_correctness.py
|
||||||
|
pytest -v -s v1/kv_connector/unit --ignore=v1/kv_connector/unit/test_multi_connector.py --ignore=v1/kv_connector/unit/test_example_connector.py --ignore=v1/kv_connector/unit/test_lmcache_integration.py --ignore=v1/kv_connector/unit/test_hf3fs_client.py --ignore=v1/kv_connector/unit/test_hf3fs_connector.py --ignore=v1/kv_connector/unit/test_hf3fs_metadata_server.py --ignore=v1/kv_connector/unit/test_offloading_connector.py
|
||||||
|
;;
|
||||||
|
server)
|
||||||
|
pip install av
|
||||||
|
cd tests
|
||||||
|
|
||||||
|
pytest -v -s entrypoints/multimodal/openai/chat_completion/test_audio_in_video.py
|
||||||
|
pytest -v -s benchmarks/test_serve_cli.py
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
echo "Unknown Intel test suite: ${test_suite}" >&2
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
@@ -243,8 +243,10 @@ container_name="xpu_${BUILDKITE_COMMIT}_$(tr -dc A-Za-z0-9 < /dev/urandom | head
|
|||||||
|
|
||||||
# ---- Command source selection ----
|
# ---- Command source selection ----
|
||||||
commands=""
|
commands=""
|
||||||
|
commands_source=""
|
||||||
if [[ -n "${VLLM_TEST_COMMANDS:-}" ]]; then
|
if [[ -n "${VLLM_TEST_COMMANDS:-}" ]]; then
|
||||||
commands="${VLLM_TEST_COMMANDS}"
|
commands="${VLLM_TEST_COMMANDS}"
|
||||||
|
commands_source="env"
|
||||||
echo "Commands sourced from VLLM_TEST_COMMANDS (quoting preserved)"
|
echo "Commands sourced from VLLM_TEST_COMMANDS (quoting preserved)"
|
||||||
elif [[ $# -gt 0 ]]; then
|
elif [[ $# -gt 0 ]]; then
|
||||||
all_yaml=true
|
all_yaml=true
|
||||||
@@ -303,8 +305,12 @@ if [[ -z "$commands" ]]; then
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
echo "Raw commands: $commands"
|
echo "Raw commands: $commands"
|
||||||
commands=$(re_quote_pytest_markers "$commands")
|
if [[ "$commands_source" != "env" ]]; then
|
||||||
echo "After re-quoting: $commands"
|
commands=$(re_quote_pytest_markers "$commands")
|
||||||
|
echo "After re-quoting: $commands"
|
||||||
|
else
|
||||||
|
echo "Skipping re-quoting for VLLM_TEST_COMMANDS input"
|
||||||
|
fi
|
||||||
commands=$(apply_intel_test_overrides "$commands")
|
commands=$(apply_intel_test_overrides "$commands")
|
||||||
echo "Final commands: $commands"
|
echo "Final commands: $commands"
|
||||||
|
|
||||||
@@ -354,7 +360,7 @@ export HF_TOKEN ZE_AFFINITY_MASK
|
|||||||
--ipc=host \
|
--ipc=host \
|
||||||
--privileged \
|
--privileged \
|
||||||
-v /dev/dri/by-path:/dev/dri/by-path \
|
-v /dev/dri/by-path:/dev/dri/by-path \
|
||||||
-v "${HOME}/.cache/huggingface:/root/.cache/huggingface" \
|
-v "/data/huggingface:/root/.cache/huggingface" \
|
||||||
--entrypoint='' \
|
--entrypoint='' \
|
||||||
-e HF_TOKEN \
|
-e HF_TOKEN \
|
||||||
-e ZE_AFFINITY_MASK \
|
-e ZE_AFFINITY_MASK \
|
||||||
@@ -363,7 +369,7 @@ export HF_TOKEN ZE_AFFINITY_MASK
|
|||||||
-e CMDS \
|
-e CMDS \
|
||||||
--name "${container_name}" \
|
--name "${container_name}" \
|
||||||
"${IMAGE}" \
|
"${IMAGE}" \
|
||||||
bash -c 'set -e; echo "ZE_AFFINITY_MASK is ${ZE_AFFINITY_MASK:-}"; eval "$CMDS"' \
|
bash -c 'set -e; source /opt/intel/oneapi/setvars.sh --force; source /opt/intel/oneapi/ccl/2021.15/env/vars.sh --force; echo "ZE_AFFINITY_MASK is ${ZE_AFFINITY_MASK:-}"; eval "$CMDS"' \
|
||||||
>/dev/null
|
>/dev/null
|
||||||
} 9>/tmp/docker-pull.lock
|
} 9>/tmp/docker-pull.lock
|
||||||
|
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ RUN pip config set global.index-url http://cache-service-vllm.nginx-pypi-cache.s
|
|||||||
|
|
||||||
# Install for pytest to make the docker build cache layer always valid
|
# Install for pytest to make the docker build cache layer always valid
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install pytest>=6.0 modelscope
|
pip install pytest>=6.0 'modelscope<1.38'
|
||||||
|
|
||||||
WORKDIR /workspace/vllm
|
WORKDIR /workspace/vllm
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,11 @@
|
|||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
if python3 -c "import torch; raise SystemExit(0 if torch.version.hip is not None else 1)"; then
|
||||||
|
uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
REQUIREMENTS_FILE="${KV_CONNECTORS_REQUIREMENTS:-/vllm-workspace/requirements/kv_connectors.txt}"
|
REQUIREMENTS_FILE="${KV_CONNECTORS_REQUIREMENTS:-/vllm-workspace/requirements/kv_connectors.txt}"
|
||||||
|
|
||||||
uv pip install --system -r "${REQUIREMENTS_FILE}"
|
uv pip install --system -r "${REQUIREMENTS_FILE}"
|
||||||
|
|||||||
@@ -130,6 +130,22 @@ docker tag public.ecr.aws/q9t5s3a7/vllm-release-repo:${ROCM_BASE_CACHE_KEY}-rocm
|
|||||||
docker push vllm/vllm-openai-rocm:latest-base
|
docker push vllm/vllm-openai-rocm:latest-base
|
||||||
docker push vllm/vllm-openai-rocm:v${RELEASE_VERSION}-base
|
docker push vllm/vllm-openai-rocm:v${RELEASE_VERSION}-base
|
||||||
|
|
||||||
|
# ---- XPU ----
|
||||||
|
|
||||||
|
docker pull public.ecr.aws/q9t5s3a7/vllm-release-repo:${COMMIT}-x86_64-xpu
|
||||||
|
|
||||||
|
docker tag public.ecr.aws/q9t5s3a7/vllm-release-repo:${COMMIT}-x86_64-xpu vllm/vllm-openai-xpu:latest-x86_64
|
||||||
|
docker tag public.ecr.aws/q9t5s3a7/vllm-release-repo:${COMMIT}-x86_64-xpu vllm/vllm-openai-xpu:v${RELEASE_VERSION}-x86_64
|
||||||
|
docker push vllm/vllm-openai-xpu:latest-x86_64
|
||||||
|
docker push vllm/vllm-openai-xpu:v${RELEASE_VERSION}-x86_64
|
||||||
|
|
||||||
|
docker manifest rm vllm/vllm-openai-xpu:latest || true
|
||||||
|
docker manifest rm vllm/vllm-openai-xpu:v${RELEASE_VERSION} || true
|
||||||
|
docker manifest create vllm/vllm-openai-xpu:latest vllm/vllm-openai-xpu:latest-x86_64 --amend
|
||||||
|
docker manifest create vllm/vllm-openai-xpu:v${RELEASE_VERSION} vllm/vllm-openai-xpu:v${RELEASE_VERSION}-x86_64 --amend
|
||||||
|
docker manifest push vllm/vllm-openai-xpu:latest
|
||||||
|
docker manifest push vllm/vllm-openai-xpu:v${RELEASE_VERSION}
|
||||||
|
|
||||||
# ---- CPU ----
|
# ---- CPU ----
|
||||||
# CPU images are behind separate block steps and may not have been built.
|
# CPU images are behind separate block steps and may not have been built.
|
||||||
# All-or-nothing: inspect both arches first, then either publish everything
|
# All-or-nothing: inspect both arches first, then either publish everything
|
||||||
|
|||||||
Executable
+32
@@ -0,0 +1,32 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Build the ROCm ci_base image, optionally from a freshly rebuilt ROCm base.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
metadata_get() {
|
||||||
|
local key="$1"
|
||||||
|
if command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
|
buildkite-agent meta-data get "${key}" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
main() {
|
||||||
|
local base_refreshed=""
|
||||||
|
|
||||||
|
base_refreshed="$(metadata_get rocm-base-refresh)"
|
||||||
|
if [[ "${base_refreshed}" == "1" ]]; then
|
||||||
|
export BASE_IMAGE
|
||||||
|
export CI_BASE_PUSH_STABLE_TAG
|
||||||
|
|
||||||
|
BASE_IMAGE="$(metadata_get rocm-base-image)"
|
||||||
|
CI_BASE_PUSH_STABLE_TAG="$(metadata_get rocm-base-push-stable-tag)"
|
||||||
|
CI_BASE_PUSH_STABLE_TAG="${CI_BASE_PUSH_STABLE_TAG:-0}"
|
||||||
|
|
||||||
|
echo "Using refreshed ROCm base image for ci_base: ${BASE_IMAGE}"
|
||||||
|
echo "Push stable ci_base tag: ${CI_BASE_PUSH_STABLE_TAG}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
bash .buildkite/scripts/ci-bake-rocm.sh ci-base-rocm-ci-with-deps
|
||||||
|
}
|
||||||
|
|
||||||
|
main "$@"
|
||||||
Executable
+57
@@ -0,0 +1,57 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Build the ROCm CI test image or wheel artifact.
|
||||||
|
#
|
||||||
|
# When Dockerfile.rocm_base changes, always build the full image so downstream
|
||||||
|
# ROCm tests can validate the freshly rebuilt base -> ci_base -> ci image chain.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
metadata_get() {
|
||||||
|
local key="$1"
|
||||||
|
if command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
|
buildkite-agent meta-data get "${key}" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
use_refreshed_base_if_present() {
|
||||||
|
local base_refreshed=""
|
||||||
|
|
||||||
|
base_refreshed="$(metadata_get rocm-base-refresh)"
|
||||||
|
if [[ "${base_refreshed}" != "1" ]]; then
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
export BASE_IMAGE
|
||||||
|
export CI_BASE_IMAGE
|
||||||
|
export IMAGE_TAG_LATEST
|
||||||
|
|
||||||
|
BASE_IMAGE="$(metadata_get rocm-base-image)"
|
||||||
|
CI_BASE_IMAGE="$(metadata_get rocm-ci-base-image)"
|
||||||
|
IMAGE_TAG_LATEST="$(metadata_get rocm-ci-image-descriptive)"
|
||||||
|
|
||||||
|
echo "Using refreshed ROCm base image for test image: ${BASE_IMAGE}"
|
||||||
|
echo "Using refreshed ROCm ci_base image for test image: ${CI_BASE_IMAGE}"
|
||||||
|
if [[ -n "${IMAGE_TAG_LATEST}" ]]; then
|
||||||
|
echo "Also tagging full ROCm CI image as: ${IMAGE_TAG_LATEST}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
|
||||||
|
main() {
|
||||||
|
local base_refreshed=0
|
||||||
|
|
||||||
|
if use_refreshed_base_if_present; then
|
||||||
|
base_refreshed=1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "${ROCM_CI_ARTIFACT_ONLY:-0}" == "1" && "${base_refreshed}" != "1" ]]; then
|
||||||
|
echo "ROCM_CI_ARTIFACT_ONLY=1; building ROCm wheel artifact only"
|
||||||
|
IMAGE_TAG="" bash .buildkite/scripts/ci-bake-rocm.sh test-rocm-ci-with-artifacts
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
|
||||||
|
bash .buildkite/scripts/ci-bake-rocm.sh test-rocm-ci-with-wheel
|
||||||
|
}
|
||||||
|
|
||||||
|
main "$@"
|
||||||
+513
@@ -0,0 +1,513 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Build and publish a fresh ROCm base image when Dockerfile.rocm_base changes.
|
||||||
|
#
|
||||||
|
# Normal AMD CI builds should not pay for this path. The script no-ops unless
|
||||||
|
# docker/Dockerfile.rocm_base changed relative to the branch base, the previous
|
||||||
|
# main commit, or ROCM_BASE_REFRESH_FORCE=1 is set.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
DOCKERFILE="${ROCM_BASE_DOCKERFILE:-docker/Dockerfile.rocm_base}"
|
||||||
|
BASE_REPO="${ROCM_BASE_IMAGE_REPO:-rocm/vllm-dev}"
|
||||||
|
CI_IMAGE_REPO="${ROCM_CI_IMAGE_REPO:-rocm/vllm-ci}"
|
||||||
|
BUILDER_NAME="${ROCM_BASE_BUILDER_NAME:-vllm-rocm-base-builder}"
|
||||||
|
DEFAULT_ROCM_BASE_METADATA_VERSION="1"
|
||||||
|
DEFAULT_ROCM_BASE_CONTENT_FILES="${DOCKERFILE}"
|
||||||
|
DEFAULT_ROCM_BASE_CONTENT_ARGS="BASE_IMAGE TRITON_BRANCH TRITON_REPO PYTORCH_BRANCH PYTORCH_REPO PYTORCH_VISION_BRANCH PYTORCH_VISION_REPO PYTORCH_AUDIO_BRANCH PYTORCH_AUDIO_REPO FA_BRANCH FA_REPO AITER_BRANCH AITER_REPO MORI_BRANCH MORI_REPO PYTORCH_ROCM_ARCH PYTHON_VERSION USE_SCCACHE"
|
||||||
|
|
||||||
|
metadata_set() {
|
||||||
|
local key="$1"
|
||||||
|
local value="$2"
|
||||||
|
|
||||||
|
[[ -n "${value}" ]] || return 0
|
||||||
|
if command -v buildkite-agent >/dev/null 2>&1; then
|
||||||
|
buildkite-agent meta-data set "${key}" "${value}" || true
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
compute_content_hash() {
|
||||||
|
local path=""
|
||||||
|
local file=""
|
||||||
|
|
||||||
|
for path in "$@"; do
|
||||||
|
if [[ -d "${path}" ]]; then
|
||||||
|
while IFS= read -r -d '' file; do
|
||||||
|
printf 'file:%s\n' "${file}"
|
||||||
|
sha256sum "${file}"
|
||||||
|
done < <(find "${path}" -type f -print0 | sort -z)
|
||||||
|
elif [[ -f "${path}" ]]; then
|
||||||
|
printf 'file:%s\n' "${path}"
|
||||||
|
sha256sum "${path}"
|
||||||
|
else
|
||||||
|
printf 'missing:%s\n' "${path}"
|
||||||
|
fi
|
||||||
|
done | sha256sum | cut -d' ' -f1
|
||||||
|
}
|
||||||
|
|
||||||
|
clean_docker_tag() {
|
||||||
|
local input="$1"
|
||||||
|
echo "${input}" | sed 's/[^a-zA-Z0-9._-]/_/g' | cut -c1-128
|
||||||
|
}
|
||||||
|
|
||||||
|
tag_component() {
|
||||||
|
local input="$1"
|
||||||
|
local max_chars="${2:-24}"
|
||||||
|
|
||||||
|
clean_docker_tag "${input:-unknown}" | cut -c1-"${max_chars}"
|
||||||
|
}
|
||||||
|
|
||||||
|
extract_arg_default() {
|
||||||
|
local arg_name="$1"
|
||||||
|
|
||||||
|
sed -n -E "s/^[[:space:]]*ARG[[:space:]]+${arg_name}=\"?([^\"[:space:]]+)\"?.*/\\1/p" \
|
||||||
|
"${DOCKERFILE}" | head -1
|
||||||
|
}
|
||||||
|
|
||||||
|
resolve_image_digest() {
|
||||||
|
local image_ref="$1"
|
||||||
|
|
||||||
|
docker buildx imagetools inspect "${image_ref}" 2>/dev/null \
|
||||||
|
| sed -n -E 's/^Digest:[[:space:]]+//p' \
|
||||||
|
| head -1 || true
|
||||||
|
}
|
||||||
|
|
||||||
|
resolve_rocm_base_arg_value() {
|
||||||
|
local arg_name="$1"
|
||||||
|
local use_sccache="$2"
|
||||||
|
|
||||||
|
case "${arg_name}" in
|
||||||
|
USE_SCCACHE)
|
||||||
|
printf '%s\n' "${use_sccache}"
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
extract_arg_default "${arg_name}"
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
|
||||||
|
hash_rocm_base_arg_values() {
|
||||||
|
local use_sccache="$1"
|
||||||
|
local base_image_digest="$2"
|
||||||
|
local arg_name=""
|
||||||
|
local arg_value=""
|
||||||
|
shift 2 || true
|
||||||
|
|
||||||
|
for arg_name in "$@"; do
|
||||||
|
[[ -n "${arg_name}" ]] || continue
|
||||||
|
arg_value=$(resolve_rocm_base_arg_value "${arg_name}" "${use_sccache}")
|
||||||
|
printf 'arg:%s=%s\n' "${arg_name}" "${arg_value:-<empty>}"
|
||||||
|
if [[ "${arg_name}" == "BASE_IMAGE" && -n "${arg_value}" ]]; then
|
||||||
|
printf 'arg:%s.digest=%s\n' "${arg_name}" "${base_image_digest:-unknown}"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
|
rocm_version_from_base_image() {
|
||||||
|
local base_image="$1"
|
||||||
|
local version=""
|
||||||
|
|
||||||
|
version="$(sed -n -E 's/.*:([0-9]+\.[0-9]+(\.[0-9]+)?)-.*/\1/p' <<<"${base_image}")"
|
||||||
|
tag_component "${version:-${base_image}}" 16
|
||||||
|
}
|
||||||
|
|
||||||
|
git_diff_changed_base() {
|
||||||
|
local range="$1"
|
||||||
|
[[ -n "$(git diff --name-only "${range}" -- "${DOCKERFILE}" 2>/dev/null)" ]]
|
||||||
|
}
|
||||||
|
|
||||||
|
short_git_ref() {
|
||||||
|
local ref="$1"
|
||||||
|
|
||||||
|
git rev-parse --short "${ref}" 2>/dev/null || printf '%s\n' "${ref}"
|
||||||
|
}
|
||||||
|
|
||||||
|
extract_arg_default_from_ref() {
|
||||||
|
local ref="$1"
|
||||||
|
local arg_name="$2"
|
||||||
|
local content=""
|
||||||
|
|
||||||
|
content="$(git show "${ref}:${DOCKERFILE}" 2>/dev/null || true)"
|
||||||
|
sed -n -E "s/^[[:space:]]*ARG[[:space:]]+${arg_name}=\"?([^\"[:space:]]+)\"?.*/\\1/p" \
|
||||||
|
<<<"${content}" | head -1
|
||||||
|
}
|
||||||
|
|
||||||
|
log_arg_default_changes() {
|
||||||
|
local old_ref="$1"
|
||||||
|
local new_ref="$2"
|
||||||
|
local content_args="${ROCM_BASE_CONTENT_ARGS:-${DEFAULT_ROCM_BASE_CONTENT_ARGS}}"
|
||||||
|
local arg_name=""
|
||||||
|
local old_value=""
|
||||||
|
local new_value=""
|
||||||
|
local changed=0
|
||||||
|
|
||||||
|
echo "Changed ROCm base ARG defaults:"
|
||||||
|
for arg_name in ${content_args}; do
|
||||||
|
old_value="$(extract_arg_default_from_ref "${old_ref}" "${arg_name}")"
|
||||||
|
new_value="$(extract_arg_default_from_ref "${new_ref}" "${arg_name}")"
|
||||||
|
if [[ "${old_value}" != "${new_value}" ]]; then
|
||||||
|
echo " - ${arg_name}: ${old_value:-<unset>} -> ${new_value:-<unset>}"
|
||||||
|
changed=1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if [[ "${changed}" == "0" ]]; then
|
||||||
|
echo " - none detected; Dockerfile instructions changed outside tracked ARG defaults"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
log_arg_line_diff() {
|
||||||
|
local range="$1"
|
||||||
|
local arg_diff=""
|
||||||
|
|
||||||
|
arg_diff="$(
|
||||||
|
git diff --unified=0 "${range}" -- "${DOCKERFILE}" 2>/dev/null \
|
||||||
|
| awk '/^[+-][[:space:]]*ARG[[:space:]]/ && $0 !~ /^(---|\+\+\+)/ { print " " $0 }' \
|
||||||
|
|| true
|
||||||
|
)"
|
||||||
|
|
||||||
|
if [[ -n "${arg_diff}" ]]; then
|
||||||
|
echo "Changed Dockerfile ARG lines:"
|
||||||
|
printf '%s\n' "${arg_diff}"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
log_rocm_base_change_check() {
|
||||||
|
local context="$1"
|
||||||
|
local range="$2"
|
||||||
|
local old_ref="$3"
|
||||||
|
local old_short=""
|
||||||
|
local head_short=""
|
||||||
|
|
||||||
|
old_short="$(short_git_ref "${old_ref}")"
|
||||||
|
head_short="$(short_git_ref HEAD)"
|
||||||
|
|
||||||
|
echo "--- :mag: ROCm base refresh check"
|
||||||
|
echo "Context: ${context}"
|
||||||
|
echo "Dockerfile: ${DOCKERFILE}"
|
||||||
|
echo "Base revision: ${old_short}"
|
||||||
|
echo "Head revision: ${head_short}"
|
||||||
|
echo "Git diff range: ${range}"
|
||||||
|
}
|
||||||
|
|
||||||
|
log_rocm_base_rebuild_reason() {
|
||||||
|
local context="$1"
|
||||||
|
local range="$2"
|
||||||
|
local old_ref="$3"
|
||||||
|
local changed_files=""
|
||||||
|
|
||||||
|
log_rocm_base_change_check "${context}" "${range}" "${old_ref}"
|
||||||
|
|
||||||
|
changed_files="$(git diff --name-only "${range}" -- "${DOCKERFILE}" 2>/dev/null || true)"
|
||||||
|
echo "Changed files:"
|
||||||
|
if [[ -n "${changed_files}" ]]; then
|
||||||
|
sed 's/^/ - /' <<<"${changed_files}"
|
||||||
|
else
|
||||||
|
echo " - ${DOCKERFILE}"
|
||||||
|
fi
|
||||||
|
log_arg_default_changes "${old_ref}" HEAD
|
||||||
|
log_arg_line_diff "${range}"
|
||||||
|
echo "Decision: rebuilding ROCm base image because ${DOCKERFILE} changed."
|
||||||
|
}
|
||||||
|
|
||||||
|
rocm_base_changed_in_range() {
|
||||||
|
local context="$1"
|
||||||
|
local range="$2"
|
||||||
|
local old_ref="$3"
|
||||||
|
|
||||||
|
if git_diff_changed_base "${range}"; then
|
||||||
|
log_rocm_base_rebuild_reason "${context}" "${range}" "${old_ref}"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
log_rocm_base_change_check "${context}" "${range}" "${old_ref}"
|
||||||
|
echo "Decision: ROCm base refresh not required; ${DOCKERFILE} is unchanged."
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
rocm_base_changed() {
|
||||||
|
local base_branch="${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}"
|
||||||
|
local base_ref="refs/remotes/origin/${base_branch}"
|
||||||
|
local merge_base=""
|
||||||
|
|
||||||
|
if [[ "${ROCM_BASE_REFRESH_SKIP:-0}" == "1" ]]; then
|
||||||
|
echo "ROCM_BASE_REFRESH_SKIP=1 set; skipping ROCm base refresh"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "${ROCM_BASE_REFRESH_FORCE:-0}" == "1" ]]; then
|
||||||
|
echo "ROCM_BASE_REFRESH_FORCE=1 set; refreshing ROCm base image"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ! git rev-parse --is-inside-work-tree >/dev/null 2>&1; then
|
||||||
|
echo "Not in a git checkout; skipping ROCm base refresh unless forced"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "${BUILDKITE_PULL_REQUEST:-false}" != "false" ]]; then
|
||||||
|
git fetch --no-tags --depth=200 origin \
|
||||||
|
"+refs/heads/${base_branch}:${base_ref}" >/dev/null 2>&1 || true
|
||||||
|
merge_base=$(git merge-base HEAD "${base_ref}" 2>/dev/null || true)
|
||||||
|
if [[ -z "${merge_base}" ]]; then
|
||||||
|
echo "Unable to determine merge base with PR base ${base_ref}; skipping ROCm base refresh unless forced"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if rocm_base_changed_in_range \
|
||||||
|
"pull request build against ${base_ref}" \
|
||||||
|
"${merge_base}...HEAD" \
|
||||||
|
"${merge_base}"; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
elif [[ "${BUILDKITE_BRANCH:-}" == "${ROCM_BASE_STABLE_BRANCH:-main}" ]] \
|
||||||
|
&& git rev-parse --verify HEAD~1 >/dev/null 2>&1; then
|
||||||
|
if rocm_base_changed_in_range \
|
||||||
|
"stable branch build; comparing against previous ${ROCM_BASE_STABLE_BRANCH:-main} commit" \
|
||||||
|
"HEAD~1..HEAD" \
|
||||||
|
"HEAD~1"; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
git fetch --no-tags --depth=200 origin \
|
||||||
|
"+refs/heads/${base_branch}:${base_ref}" >/dev/null 2>&1 || true
|
||||||
|
merge_base=$(git merge-base HEAD "${base_ref}" 2>/dev/null || true)
|
||||||
|
if [[ -z "${merge_base}" ]]; then
|
||||||
|
echo "Unable to determine merge base with branch base ${base_ref}; skipping ROCm base refresh unless forced"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if rocm_base_changed_in_range \
|
||||||
|
"branch build against ${base_ref}" \
|
||||||
|
"${merge_base}...HEAD" \
|
||||||
|
"${merge_base}"; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
should_push_stable_tag() {
|
||||||
|
if [[ "${BUILDKITE_PULL_REQUEST:-false}" != "false" ]]; then
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "${ROCM_BASE_PUSH_STABLE_TAG:-}" == "1" ]]; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
if [[ "${ROCM_BASE_PUSH_STABLE_TAG:-}" == "0" ]]; then
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
[[ "${BUILDKITE_PULL_REQUEST:-false}" == "false" \
|
||||||
|
&& "${BUILDKITE_BRANCH:-}" == "${ROCM_BASE_STABLE_BRANCH:-main}" ]]
|
||||||
|
}
|
||||||
|
|
||||||
|
setup_builder() {
|
||||||
|
echo "--- :buildkite: Setting up buildx builder for ROCm base"
|
||||||
|
if docker buildx inspect "${BUILDER_NAME}" >/dev/null 2>&1; then
|
||||||
|
docker buildx use "${BUILDER_NAME}"
|
||||||
|
else
|
||||||
|
docker buildx create --name "${BUILDER_NAME}" --driver docker-container --use
|
||||||
|
fi
|
||||||
|
docker buildx inspect --bootstrap
|
||||||
|
}
|
||||||
|
|
||||||
|
compute_base_content_hash() {
|
||||||
|
local use_sccache="$1"
|
||||||
|
local base_image_digest="$2"
|
||||||
|
local content_files="${ROCM_BASE_CONTENT_FILES:-${DEFAULT_ROCM_BASE_CONTENT_FILES}}"
|
||||||
|
local content_args="${ROCM_BASE_CONTENT_ARGS:-${DEFAULT_ROCM_BASE_CONTENT_ARGS}}"
|
||||||
|
local -a content_paths=()
|
||||||
|
local -a content_arg_names=()
|
||||||
|
|
||||||
|
read -r -a content_paths <<< "${content_files}"
|
||||||
|
read -r -a content_arg_names <<< "${content_args}"
|
||||||
|
|
||||||
|
{
|
||||||
|
printf 'content-files-hash:%s\n' "$(compute_content_hash "${content_paths[@]}")"
|
||||||
|
printf 'dockerfile:%s\n' "${DOCKERFILE}"
|
||||||
|
printf 'resolved-build-args:\n'
|
||||||
|
hash_rocm_base_arg_values \
|
||||||
|
"${use_sccache}" "${base_image_digest}" "${content_arg_names[@]}"
|
||||||
|
} | sha256sum | cut -d' ' -f1
|
||||||
|
}
|
||||||
|
|
||||||
|
build_base_image() {
|
||||||
|
local use_sccache="${ROCM_BASE_USE_SCCACHE:-${USE_SCCACHE:-0}}"
|
||||||
|
local base_hash=""
|
||||||
|
local build_date=""
|
||||||
|
local build_suffix=""
|
||||||
|
local base_image_arg=""
|
||||||
|
local base_image_digest=""
|
||||||
|
local rocm_version=""
|
||||||
|
local triton_arg=""
|
||||||
|
local pytorch_arg=""
|
||||||
|
local pytorch_vision_arg=""
|
||||||
|
local pytorch_audio_arg=""
|
||||||
|
local fa_arg=""
|
||||||
|
local aiter_arg=""
|
||||||
|
local mori_arg=""
|
||||||
|
local python_version_arg=""
|
||||||
|
local pytorch_rocm_arch_arg=""
|
||||||
|
local pytorch_branch=""
|
||||||
|
local aiter_branch=""
|
||||||
|
local dependency_summary=""
|
||||||
|
local descriptor=""
|
||||||
|
local ci_descriptor=""
|
||||||
|
local descriptive_tag=""
|
||||||
|
local stable_tag="${BASE_REPO}:base"
|
||||||
|
local ci_descriptive_tag=""
|
||||||
|
local content_files="${ROCM_BASE_CONTENT_FILES:-${DEFAULT_ROCM_BASE_CONTENT_FILES}}"
|
||||||
|
local content_args="${ROCM_BASE_CONTENT_ARGS:-${DEFAULT_ROCM_BASE_CONTENT_ARGS}}"
|
||||||
|
local content_files_hash=""
|
||||||
|
local metadata_version="${ROCM_BASE_METADATA_VERSION:-${DEFAULT_ROCM_BASE_METADATA_VERSION}}"
|
||||||
|
local -a tags=()
|
||||||
|
local -a no_cache_args=()
|
||||||
|
local -a sccache_args=()
|
||||||
|
local -a content_paths=()
|
||||||
|
|
||||||
|
if [[ ! -f "${DOCKERFILE}" ]]; then
|
||||||
|
echo "Error: ROCm base Dockerfile not found: ${DOCKERFILE}" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
build_date="${ROCM_BASE_TAG_DATE:-$(date -u +%Y%m%d)}"
|
||||||
|
if [[ -n "${BUILDKITE_BUILD_NUMBER:-}" ]]; then
|
||||||
|
build_suffix="_bk_${BUILDKITE_BUILD_NUMBER}"
|
||||||
|
fi
|
||||||
|
base_image_arg="$(extract_arg_default BASE_IMAGE)"
|
||||||
|
base_image_digest="$(resolve_image_digest "${base_image_arg}")"
|
||||||
|
read -r -a content_paths <<< "${content_files}"
|
||||||
|
content_files_hash="$(compute_content_hash "${content_paths[@]}")"
|
||||||
|
base_hash=$(compute_base_content_hash "${use_sccache}" "${base_image_digest}")
|
||||||
|
rocm_version="$(rocm_version_from_base_image "${base_image_arg}")"
|
||||||
|
triton_arg="$(extract_arg_default TRITON_BRANCH)"
|
||||||
|
pytorch_arg="$(extract_arg_default PYTORCH_BRANCH)"
|
||||||
|
pytorch_vision_arg="$(extract_arg_default PYTORCH_VISION_BRANCH)"
|
||||||
|
pytorch_audio_arg="$(extract_arg_default PYTORCH_AUDIO_BRANCH)"
|
||||||
|
fa_arg="$(extract_arg_default FA_BRANCH)"
|
||||||
|
aiter_arg="$(extract_arg_default AITER_BRANCH)"
|
||||||
|
mori_arg="$(extract_arg_default MORI_BRANCH)"
|
||||||
|
python_version_arg="$(extract_arg_default PYTHON_VERSION)"
|
||||||
|
pytorch_rocm_arch_arg="$(extract_arg_default PYTORCH_ROCM_ARCH)"
|
||||||
|
pytorch_branch="$(tag_component "${pytorch_arg}" 16)"
|
||||||
|
aiter_branch="$(tag_component "${aiter_arg}" 24)"
|
||||||
|
dependency_summary="base=${base_image_arg},rocm=${rocm_version},python=${python_version_arg},pytorch=${pytorch_arg},torchvision=${pytorch_vision_arg},torchaudio=${pytorch_audio_arg},triton=${triton_arg},flash-attn=${fa_arg},aiter=${aiter_arg},mori=${mori_arg},pytorch-rocm-arch=${pytorch_rocm_arch_arg}"
|
||||||
|
descriptor="$(clean_docker_tag "base_custom_aiter_${aiter_branch}_torch_${pytorch_branch}_${build_date}${build_suffix}")"
|
||||||
|
ci_descriptor="$(clean_docker_tag "ci_custom_aiter_${aiter_branch}_torch_${pytorch_branch}_${build_date}${build_suffix}")"
|
||||||
|
|
||||||
|
descriptive_tag="${BASE_REPO}:${descriptor}"
|
||||||
|
ci_descriptive_tag="${CI_IMAGE_REPO}:${ci_descriptor}"
|
||||||
|
|
||||||
|
tags=(-t "${descriptive_tag}")
|
||||||
|
if should_push_stable_tag; then
|
||||||
|
tags+=(-t "${stable_tag}")
|
||||||
|
metadata_set "rocm-base-push-stable-tag" "1"
|
||||||
|
else
|
||||||
|
metadata_set "rocm-base-push-stable-tag" "0"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "${ROCM_BASE_NO_CACHE:-1}" == "1" ]]; then
|
||||||
|
no_cache_args=(--no-cache)
|
||||||
|
fi
|
||||||
|
|
||||||
|
for env_name in \
|
||||||
|
SCCACHE_DOWNLOAD_URL \
|
||||||
|
SCCACHE_ENDPOINT \
|
||||||
|
SCCACHE_BUCKET_NAME \
|
||||||
|
SCCACHE_REGION_NAME \
|
||||||
|
SCCACHE_S3_NO_CREDENTIALS; do
|
||||||
|
if [[ -n "${!env_name:-}" ]]; then
|
||||||
|
sccache_args+=(--build-arg "${env_name}=${!env_name}")
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "--- :docker: Building ROCm base image"
|
||||||
|
echo "Dockerfile: ${DOCKERFILE}"
|
||||||
|
echo "Descriptive tag: ${descriptive_tag}"
|
||||||
|
echo "Stable tag: ${stable_tag} ($(should_push_stable_tag && echo enabled || echo disabled))"
|
||||||
|
echo "Content hash: ${base_hash}"
|
||||||
|
echo "Dependency summary: ${dependency_summary}"
|
||||||
|
echo "USE_SCCACHE: ${use_sccache}"
|
||||||
|
|
||||||
|
docker buildx build \
|
||||||
|
"${no_cache_args[@]}" \
|
||||||
|
--pull \
|
||||||
|
--progress "${BUILDKIT_PROGRESS:-plain}" \
|
||||||
|
--file "${DOCKERFILE}" \
|
||||||
|
--build-arg "USE_SCCACHE=${use_sccache}" \
|
||||||
|
"${sccache_args[@]}" \
|
||||||
|
--label "org.opencontainers.image.source=https://github.com/vllm-project/vllm" \
|
||||||
|
--label "org.opencontainers.image.vendor=vLLM" \
|
||||||
|
--label "org.opencontainers.image.title=vLLM ROCm base" \
|
||||||
|
--label "org.opencontainers.image.revision=${BUILDKITE_COMMIT:-}" \
|
||||||
|
--label "vllm.rocm_base.metadata_version=${metadata_version}" \
|
||||||
|
--label "vllm.rocm_base.content_hash=${base_hash}" \
|
||||||
|
--label "vllm.rocm_base.content_files_hash=${content_files_hash}" \
|
||||||
|
--label "vllm.rocm_base.dockerfile=${DOCKERFILE}" \
|
||||||
|
--label "vllm.rocm_base.image.descriptive=${descriptive_tag}" \
|
||||||
|
--label "vllm.rocm_base.image.stable=${stable_tag}" \
|
||||||
|
--label "vllm.rocm_base.git_commit=${BUILDKITE_COMMIT:-}" \
|
||||||
|
--label "vllm.rocm_base.stable_branch=${ROCM_BASE_STABLE_BRANCH:-main}" \
|
||||||
|
--label "vllm.rocm_base.descriptor=${descriptor}" \
|
||||||
|
--label "vllm.rocm_base.dependency_summary=${dependency_summary}" \
|
||||||
|
--label "vllm.rocm_base.base_image=${base_image_arg}" \
|
||||||
|
--label "vllm.rocm_base.base_image_digest=${base_image_digest}" \
|
||||||
|
--label "vllm.rocm_base.dependency.rocm=${rocm_version}" \
|
||||||
|
--label "vllm.rocm_base.dependency.python=${python_version_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.pytorch=${pytorch_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.torchvision=${pytorch_vision_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.torchaudio=${pytorch_audio_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.triton=${triton_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.flash_attention=${fa_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.aiter=${aiter_arg}" \
|
||||||
|
--label "vllm.rocm_base.dependency.mori=${mori_arg}" \
|
||||||
|
--label "vllm.rocm_base.pytorch_rocm_arch=${pytorch_rocm_arch_arg}" \
|
||||||
|
"${tags[@]}" \
|
||||||
|
--push \
|
||||||
|
.
|
||||||
|
|
||||||
|
docker buildx imagetools inspect "${descriptive_tag}" >/dev/null
|
||||||
|
|
||||||
|
metadata_set "rocm-base-refresh" "1"
|
||||||
|
metadata_set "rocm-base-image" "${descriptive_tag}"
|
||||||
|
metadata_set "rocm-base-image-descriptive" "${descriptive_tag}"
|
||||||
|
metadata_set "rocm-base-image-stable" "${stable_tag}"
|
||||||
|
metadata_set "rocm-base-image-ci-descriptive" "${ci_descriptive_tag}"
|
||||||
|
metadata_set "rocm-base-metadata-version" "${metadata_version}"
|
||||||
|
metadata_set "rocm-base-content-hash" "${base_hash}"
|
||||||
|
metadata_set "rocm-base-content-files-hash" "${content_files_hash}"
|
||||||
|
metadata_set "rocm-base-content-files" "${content_files}"
|
||||||
|
metadata_set "rocm-base-content-args" "${content_args}"
|
||||||
|
metadata_set "rocm-base-base-image-digest" "${base_image_digest}"
|
||||||
|
metadata_set "rocm-base-dockerfile" "${DOCKERFILE}"
|
||||||
|
metadata_set "rocm-base-descriptor" "${descriptor}"
|
||||||
|
metadata_set "rocm-base-dependency-summary" "${dependency_summary}"
|
||||||
|
metadata_set "rocm-base-dependency-rocm" "${rocm_version}"
|
||||||
|
metadata_set "rocm-base-dependency-python" "${python_version_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-pytorch" "${pytorch_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-torchvision" "${pytorch_vision_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-torchaudio" "${pytorch_audio_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-triton" "${triton_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-flash-attention" "${fa_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-aiter" "${aiter_arg}"
|
||||||
|
metadata_set "rocm-base-dependency-mori" "${mori_arg}"
|
||||||
|
metadata_set "rocm-base-pytorch-rocm-arch" "${pytorch_rocm_arch_arg}"
|
||||||
|
metadata_set "rocm-ci-image-descriptive" "${ci_descriptive_tag}"
|
||||||
|
|
||||||
|
echo "--- :white_check_mark: ROCm base image published"
|
||||||
|
echo "Use BASE_IMAGE=${descriptive_tag} for downstream ROCm CI builds"
|
||||||
|
}
|
||||||
|
|
||||||
|
main() {
|
||||||
|
metadata_set "rocm-base-refresh" "0"
|
||||||
|
|
||||||
|
if ! rocm_base_changed; then
|
||||||
|
echo "ROCm base Dockerfile did not change; skipping base image refresh"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
setup_builder
|
||||||
|
build_base_image
|
||||||
|
}
|
||||||
|
|
||||||
|
main "$@"
|
||||||
Executable
+32
@@ -0,0 +1,32 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Fast structural smoke test for the full ROCm CI image.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
image_ref="${VLLM_CI_SMOKE_IMAGE:-rocm/vllm-ci:${BUILDKITE_COMMIT:?BUILDKITE_COMMIT is required}}"
|
||||||
|
|
||||||
|
docker run --rm --network=none --entrypoint /bin/bash "${image_ref}" -ec '
|
||||||
|
if [ ! -d /vllm-workspace ]; then echo Missing directory: /vllm-workspace >&2; exit 1; fi
|
||||||
|
if [ ! -d /vllm-workspace/tests ]; then echo Missing directory: /vllm-workspace/tests >&2; exit 1; fi
|
||||||
|
if [ ! -d /vllm-workspace/src/vllm ]; then echo Missing directory: /vllm-workspace/src/vllm >&2; exit 1; fi
|
||||||
|
if [ ! -x /vllm-workspace/src/vllm/vllm-rs ]; then echo Missing executable: /vllm-workspace/src/vllm/vllm-rs >&2; exit 1; fi
|
||||||
|
|
||||||
|
command -v python3
|
||||||
|
command -v uv
|
||||||
|
command -v pytest
|
||||||
|
|
||||||
|
if ! command -v amd-smi >/dev/null 2>&1 && ! command -v rocminfo >/dev/null 2>&1; then
|
||||||
|
echo No ROCm CLI found in image >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
python3 - <<PY
|
||||||
|
import torch
|
||||||
|
import vllm
|
||||||
|
|
||||||
|
print(torch.__version__)
|
||||||
|
print(vllm.__version__)
|
||||||
|
PY
|
||||||
|
|
||||||
|
echo AMD image smoke OK
|
||||||
|
'
|
||||||
@@ -109,7 +109,9 @@ run_nodes() {
|
|||||||
if [ "$node" -ne 0 ]; then
|
if [ "$node" -ne 0 ]; then
|
||||||
docker exec -d "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
docker exec -d "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
||||||
else
|
else
|
||||||
docker exec "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
# Allocate a TTY (-t -i) for the foreground head node so its output
|
||||||
|
# keeps ANSI color in the Buildkite log (see run-amd-test.sh).
|
||||||
|
docker exec -t -i "node$node" /bin/bash -c "cd $WORKING_DIR ; ${COMMANDS[$node]}"
|
||||||
fi
|
fi
|
||||||
done
|
done
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -8,7 +8,12 @@ if [[ "$MODE" != "style-clippy" && "$MODE" != "test" ]]; then
|
|||||||
exit 2
|
exit 2
|
||||||
fi
|
fi
|
||||||
|
|
||||||
ROOT_DIR="$(git rev-parse --show-toplevel)"
|
if ROOT_DIR="$(git rev-parse --show-toplevel 2>/dev/null)"; then
|
||||||
|
:
|
||||||
|
else
|
||||||
|
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd -P)"
|
||||||
|
ROOT_DIR="$(cd -- "${SCRIPT_DIR}/../.." && pwd -P)"
|
||||||
|
fi
|
||||||
cd "$ROOT_DIR"
|
cd "$ROOT_DIR"
|
||||||
|
|
||||||
export CARGO_TERM_COLOR="${CARGO_TERM_COLOR:-always}"
|
export CARGO_TERM_COLOR="${CARGO_TERM_COLOR:-always}"
|
||||||
@@ -16,16 +21,20 @@ export CARGO_HOME="${CARGO_HOME:-$HOME/.cargo}"
|
|||||||
export RUSTUP_HOME="${RUSTUP_HOME:-$HOME/.rustup}"
|
export RUSTUP_HOME="${RUSTUP_HOME:-$HOME/.rustup}"
|
||||||
export PATH="$CARGO_HOME/bin:$PATH"
|
export PATH="$CARGO_HOME/bin:$PATH"
|
||||||
|
|
||||||
|
PROTOC_VERSION="${PROTOC_VERSION:-31.1}"
|
||||||
|
CARGO_BINSTALL_VERSION="${CARGO_BINSTALL_VERSION:-1.20.1}"
|
||||||
|
UV_VERSION="${UV_VERSION:-0.11.28}"
|
||||||
|
PYO3_PYTHON_VERSION="${PYO3_PYTHON_VERSION:-3.12}"
|
||||||
|
|
||||||
|
CARGO_SORT_VERSION_REQ="${CARGO_SORT_VERSION_REQ:-2}"
|
||||||
|
CARGO_DENY_VERSION_REQ="${CARGO_DENY_VERSION_REQ:-0.20}"
|
||||||
|
CARGO_NEXTEST_VERSION_REQ="${CARGO_NEXTEST_VERSION_REQ:-0.9}"
|
||||||
|
|
||||||
log_section() {
|
log_section() {
|
||||||
echo "--- $*"
|
echo "--- $*"
|
||||||
}
|
}
|
||||||
|
|
||||||
install_protoc() {
|
install_protoc() {
|
||||||
if command -v protoc >/dev/null 2>&1; then
|
|
||||||
return
|
|
||||||
fi
|
|
||||||
|
|
||||||
local version="${PROTOC_VERSION:-31.1}"
|
|
||||||
local arch
|
local arch
|
||||||
case "$(uname -m)" in
|
case "$(uname -m)" in
|
||||||
x86_64)
|
x86_64)
|
||||||
@@ -40,16 +49,17 @@ install_protoc() {
|
|||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
local url="https://github.com/protocolbuffers/protobuf/releases/download/v${version}/protoc-${version}-linux-${arch}.zip"
|
local url="https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOC_VERSION}/protoc-${PROTOC_VERSION}-linux-${arch}.zip"
|
||||||
local tmp_dir
|
local tmp_dir
|
||||||
tmp_dir="$(mktemp -d)"
|
tmp_dir="$(mktemp -d)"
|
||||||
|
|
||||||
log_section "Installing protoc ${version}"
|
log_section "Installing protoc ${PROTOC_VERSION}"
|
||||||
curl -L --proto '=https' --tlsv1.2 -sSf "$url" -o "$tmp_dir/protoc.zip"
|
curl -L --proto '=https' --tlsv1.2 -sSf "$url" -o "$tmp_dir/protoc.zip"
|
||||||
mkdir -p "$CARGO_HOME/bin"
|
mkdir -p "$CARGO_HOME/bin"
|
||||||
unzip -q "$tmp_dir/protoc.zip" bin/protoc 'include/*' -d "$CARGO_HOME"
|
unzip -q "$tmp_dir/protoc.zip" bin/protoc 'include/*' -d "$CARGO_HOME"
|
||||||
chmod +x "$CARGO_HOME/bin/protoc"
|
chmod +x "$CARGO_HOME/bin/protoc"
|
||||||
rm -rf "$tmp_dir"
|
rm -rf "$tmp_dir"
|
||||||
|
protoc --version
|
||||||
}
|
}
|
||||||
|
|
||||||
rust_toolchain() {
|
rust_toolchain() {
|
||||||
@@ -70,56 +80,48 @@ install_rust_toolchain() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
install_cargo_binstall() {
|
install_cargo_binstall() {
|
||||||
if command -v cargo-binstall >/dev/null 2>&1; then
|
log_section "Installing cargo-binstall ${CARGO_BINSTALL_VERSION}"
|
||||||
return
|
|
||||||
fi
|
|
||||||
|
|
||||||
log_section "Installing cargo-binstall"
|
|
||||||
curl -L --proto '=https' --tlsv1.2 -sSf \
|
curl -L --proto '=https' --tlsv1.2 -sSf \
|
||||||
https://raw.githubusercontent.com/cargo-bins/cargo-binstall/main/install-from-binstall-release.sh \
|
"https://raw.githubusercontent.com/cargo-bins/cargo-binstall/v${CARGO_BINSTALL_VERSION}/install-from-binstall-release.sh" \
|
||||||
| bash
|
| env BINSTALL_VERSION="$CARGO_BINSTALL_VERSION" bash
|
||||||
|
cargo-binstall -V
|
||||||
}
|
}
|
||||||
|
|
||||||
install_cargo_sort() {
|
install_cargo_sort() {
|
||||||
if command -v cargo-sort >/dev/null 2>&1; then
|
log_section "Installing cargo-sort ${CARGO_SORT_VERSION_REQ}"
|
||||||
return
|
cargo binstall --no-confirm --force "cargo-sort@${CARGO_SORT_VERSION_REQ}"
|
||||||
fi
|
}
|
||||||
|
|
||||||
log_section "Installing cargo-sort"
|
install_cargo_deny() {
|
||||||
install_cargo_binstall
|
log_section "Installing cargo-deny ${CARGO_DENY_VERSION_REQ}"
|
||||||
cargo binstall --no-confirm cargo-sort
|
cargo binstall --no-confirm --force "cargo-deny@${CARGO_DENY_VERSION_REQ}"
|
||||||
}
|
}
|
||||||
|
|
||||||
install_cargo_nextest() {
|
install_cargo_nextest() {
|
||||||
if command -v cargo-nextest >/dev/null 2>&1; then
|
log_section "Installing cargo-nextest ${CARGO_NEXTEST_VERSION_REQ}"
|
||||||
return
|
cargo binstall \
|
||||||
fi
|
--no-confirm \
|
||||||
|
--force \
|
||||||
log_section "Installing cargo-nextest"
|
--secure \
|
||||||
install_cargo_binstall
|
"cargo-nextest@${CARGO_NEXTEST_VERSION_REQ}"
|
||||||
cargo binstall --no-confirm --secure cargo-nextest
|
|
||||||
}
|
}
|
||||||
|
|
||||||
install_uv() {
|
install_uv() {
|
||||||
if command -v uv >/dev/null 2>&1; then
|
log_section "Installing uv ${UV_VERSION}"
|
||||||
return
|
curl -L --proto '=https' --tlsv1.2 -sSf \
|
||||||
fi
|
"https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-installer.sh" \
|
||||||
|
|
||||||
log_section "Installing uv"
|
|
||||||
curl -LsSf --proto '=https' --tlsv1.2 https://astral.sh/uv/install.sh \
|
|
||||||
| env UV_INSTALL_DIR="$CARGO_HOME/bin" sh
|
| env UV_INSTALL_DIR="$CARGO_HOME/bin" sh
|
||||||
|
uv --version
|
||||||
}
|
}
|
||||||
|
|
||||||
setup_pyo3_python() {
|
setup_pyo3_python() {
|
||||||
local python_version="${PYO3_PYTHON_VERSION:-3.12}"
|
log_section "Installing Python ${PYO3_PYTHON_VERSION} for PyO3 tests"
|
||||||
|
uv python install "$PYO3_PYTHON_VERSION"
|
||||||
log_section "Installing Python ${python_version} for PyO3 tests"
|
|
||||||
uv python install "$python_version"
|
|
||||||
PYO3_PYTHON="$(uv python find \
|
PYO3_PYTHON="$(uv python find \
|
||||||
--managed-python \
|
--managed-python \
|
||||||
--no-project \
|
--no-project \
|
||||||
--resolve-links \
|
--resolve-links \
|
||||||
"$python_version")"
|
"$PYO3_PYTHON_VERSION")"
|
||||||
export PYO3_PYTHON
|
export PYO3_PYTHON
|
||||||
|
|
||||||
local python_libdir
|
local python_libdir
|
||||||
@@ -141,7 +143,9 @@ PY
|
|||||||
}
|
}
|
||||||
|
|
||||||
run_style_clippy() {
|
run_style_clippy() {
|
||||||
|
install_cargo_binstall
|
||||||
install_cargo_sort
|
install_cargo_sort
|
||||||
|
install_cargo_deny
|
||||||
|
|
||||||
log_section "Checking Rust formatting"
|
log_section "Checking Rust formatting"
|
||||||
cargo fmt --manifest-path rust/Cargo.toml --all -- --check
|
cargo fmt --manifest-path rust/Cargo.toml --all -- --check
|
||||||
@@ -149,6 +153,13 @@ run_style_clippy() {
|
|||||||
log_section "Checking Cargo.toml ordering"
|
log_section "Checking Cargo.toml ordering"
|
||||||
cargo sort --workspace --check rust
|
cargo sort --workspace --check rust
|
||||||
|
|
||||||
|
log_section "Checking Rust dependency bans"
|
||||||
|
cargo deny \
|
||||||
|
--manifest-path rust/Cargo.toml \
|
||||||
|
--config rust/deny.toml \
|
||||||
|
check \
|
||||||
|
bans
|
||||||
|
|
||||||
log_section "Running clippy"
|
log_section "Running clippy"
|
||||||
cargo clippy \
|
cargo clippy \
|
||||||
--manifest-path rust/Cargo.toml \
|
--manifest-path rust/Cargo.toml \
|
||||||
@@ -163,6 +174,7 @@ run_style_clippy() {
|
|||||||
run_tests() {
|
run_tests() {
|
||||||
install_uv
|
install_uv
|
||||||
setup_pyo3_python
|
setup_pyo3_python
|
||||||
|
install_cargo_binstall
|
||||||
install_cargo_nextest
|
install_cargo_nextest
|
||||||
|
|
||||||
log_section "Running cargo nextest"
|
log_section "Running cargo nextest"
|
||||||
|
|||||||
@@ -33,6 +33,14 @@ if [[ -n "${ATTENTION_BACKEND:-}" ]]; then
|
|||||||
EXTRA_ARGS+=(--attention-backend "${ATTENTION_BACKEND}")
|
EXTRA_ARGS+=(--attention-backend "${ATTENTION_BACKEND}")
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# ROCm: run eager to avoid intermittent HIP-graph decode corruption.
|
||||||
|
# See https://github.com/ROCm/clr/issues/279
|
||||||
|
# TODO(aarushjain29): Revert after TheRock 7.14
|
||||||
|
if command -v rocm-smi &> /dev/null || command -v amd-smi &> /dev/null || [[ -d /opt/rocm ]] || [[ -n "${ROCM_PATH:-}" ]]; then
|
||||||
|
echo "ROCm platform detected: adding --enforce-eager to avoid HIP-graph decode corruption"
|
||||||
|
EXTRA_ARGS+=(--enforce-eager)
|
||||||
|
fi
|
||||||
|
|
||||||
cleanup() {
|
cleanup() {
|
||||||
if [[ -n "${SERVER_PID:-}" ]] && kill -0 "${SERVER_PID}" 2>/dev/null; then
|
if [[ -n "${SERVER_PID:-}" ]] && kill -0 "${SERVER_PID}" 2>/dev/null; then
|
||||||
kill "${SERVER_PID}" 2>/dev/null || true
|
kill "${SERVER_PID}" 2>/dev/null || true
|
||||||
|
|||||||
@@ -18,6 +18,10 @@ wait_for_server() {
|
|||||||
|
|
||||||
MODEL="Qwen/Qwen3-30B-A3B-FP8"
|
MODEL="Qwen/Qwen3-30B-A3B-FP8"
|
||||||
BACK="allgather_reducescatter"
|
BACK="allgather_reducescatter"
|
||||||
|
if command -v rocm-smi &> /dev/null || [[ -d /opt/rocm ]] || [[ -n "${ROCM_PATH:-}" ]]; then
|
||||||
|
# Disable MOE padding for ROCm since it is causing eplb to fail.
|
||||||
|
export VLLM_ROCM_MOE_PADDING=0
|
||||||
|
fi
|
||||||
|
|
||||||
cleanup() {
|
cleanup() {
|
||||||
if [[ -n "${SERVER_PID:-}" ]] && kill -0 "${SERVER_PID}" 2>/dev/null; then
|
if [[ -n "${SERVER_PID:-}" ]] && kill -0 "${SERVER_PID}" 2>/dev/null; then
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ NC='\033[0m' # No Color
|
|||||||
# Default configuration
|
# Default configuration
|
||||||
PIPELINE="ci"
|
PIPELINE="ci"
|
||||||
DRY_RUN=true
|
DRY_RUN=true
|
||||||
|
TORCH_NIGHTLY=false
|
||||||
|
|
||||||
usage() {
|
usage() {
|
||||||
cat <<EOF
|
cat <<EOF
|
||||||
@@ -34,12 +35,14 @@ Sets RUN_ALL=1 and NIGHTLY=1 environment variables.
|
|||||||
SAFETY: Dry-run by default. Use --execute to actually trigger a build.
|
SAFETY: Dry-run by default. Use --execute to actually trigger a build.
|
||||||
|
|
||||||
Options:
|
Options:
|
||||||
--execute Actually trigger the build (default: dry-run)
|
--execute Actually trigger the build (default: dry-run)
|
||||||
--pipeline Buildkite pipeline slug (default: ${PIPELINE})
|
--pipeline Buildkite pipeline slug (default: ${PIPELINE})
|
||||||
--commit Override commit SHA (default: current HEAD)
|
--commit Override commit SHA (default: current HEAD)
|
||||||
--branch Override branch name (default: current branch)
|
--branch Override branch name (default: current branch)
|
||||||
--message Custom build message (default: auto-generated)
|
--message Custom build message (default: auto-generated)
|
||||||
--help Show this help message
|
--torch-nightly Also build and run the full suite against torch nightly
|
||||||
|
(sets TORCH_NIGHTLY=1)
|
||||||
|
--help Show this help message
|
||||||
|
|
||||||
Prerequisites:
|
Prerequisites:
|
||||||
- bk CLI installed: brew tap buildkite/buildkite && brew install buildkite/buildkite/bk
|
- bk CLI installed: brew tap buildkite/buildkite && brew install buildkite/buildkite/bk
|
||||||
@@ -49,6 +52,7 @@ Examples:
|
|||||||
$(basename "$0") # Dry-run, show what would happen
|
$(basename "$0") # Dry-run, show what would happen
|
||||||
$(basename "$0") --execute # Actually trigger the build
|
$(basename "$0") --execute # Actually trigger the build
|
||||||
$(basename "$0") --pipeline ci-shadow # Dry-run with different pipeline
|
$(basename "$0") --pipeline ci-shadow # Dry-run with different pipeline
|
||||||
|
$(basename "$0") --torch-nightly # Dry-run a full torch-nightly run
|
||||||
EOF
|
EOF
|
||||||
exit 1
|
exit 1
|
||||||
}
|
}
|
||||||
@@ -96,6 +100,10 @@ while [[ $# -gt 0 ]]; do
|
|||||||
MESSAGE="$2"
|
MESSAGE="$2"
|
||||||
shift 2
|
shift 2
|
||||||
;;
|
;;
|
||||||
|
--torch-nightly)
|
||||||
|
TORCH_NIGHTLY=true
|
||||||
|
shift
|
||||||
|
;;
|
||||||
--help|-h)
|
--help|-h)
|
||||||
usage
|
usage
|
||||||
;;
|
;;
|
||||||
@@ -171,11 +179,17 @@ if [[ $(echo "$REMOTE_BRANCHES" | wc -l) -gt 5 ]]; then
|
|||||||
fi
|
fi
|
||||||
echo ""
|
echo ""
|
||||||
|
|
||||||
|
# Environment variables passed to the build.
|
||||||
|
BUILD_ENV=("RUN_ALL=1" "NIGHTLY=1")
|
||||||
|
if [[ "$TORCH_NIGHTLY" == true ]]; then
|
||||||
|
BUILD_ENV+=("TORCH_NIGHTLY=1")
|
||||||
|
fi
|
||||||
|
|
||||||
log_info "Pipeline: ${PIPELINE}"
|
log_info "Pipeline: ${PIPELINE}"
|
||||||
log_info "Branch: ${BRANCH}"
|
log_info "Branch: ${BRANCH}"
|
||||||
log_info "Commit: ${COMMIT}"
|
log_info "Commit: ${COMMIT}"
|
||||||
log_info "Message: ${MESSAGE}"
|
log_info "Message: ${MESSAGE}"
|
||||||
log_info "Environment: RUN_ALL=1, NIGHTLY=1"
|
log_info "Environment: ${BUILD_ENV[*]}"
|
||||||
echo ""
|
echo ""
|
||||||
|
|
||||||
# Build the command
|
# Build the command
|
||||||
@@ -187,9 +201,10 @@ CMD=(bk build create
|
|||||||
--commit "${COMMIT}"
|
--commit "${COMMIT}"
|
||||||
--branch "${BRANCH}"
|
--branch "${BRANCH}"
|
||||||
--message "${MESSAGE}"
|
--message "${MESSAGE}"
|
||||||
--env "RUN_ALL=1"
|
|
||||||
--env "NIGHTLY=1"
|
|
||||||
)
|
)
|
||||||
|
for env_var in "${BUILD_ENV[@]}"; do
|
||||||
|
CMD+=(--env "${env_var}")
|
||||||
|
done
|
||||||
|
|
||||||
if [[ "$DRY_RUN" == true ]]; then
|
if [[ "$DRY_RUN" == true ]]; then
|
||||||
echo "=========================================="
|
echo "=========================================="
|
||||||
@@ -210,8 +225,14 @@ if [[ "$DRY_RUN" == true ]]; then
|
|||||||
echo " --commit '$(escape_for_shell "${COMMIT}")' \\"
|
echo " --commit '$(escape_for_shell "${COMMIT}")' \\"
|
||||||
echo " --branch '$(escape_for_shell "${BRANCH}")' \\"
|
echo " --branch '$(escape_for_shell "${BRANCH}")' \\"
|
||||||
echo " --message '$(escape_for_shell "${MESSAGE}")' \\"
|
echo " --message '$(escape_for_shell "${MESSAGE}")' \\"
|
||||||
echo " --env 'RUN_ALL=1' \\"
|
last_idx=$(( ${#BUILD_ENV[@]} - 1 ))
|
||||||
echo " --env 'NIGHTLY=1'"
|
for i in "${!BUILD_ENV[@]}"; do
|
||||||
|
if [[ $i -eq $last_idx ]]; then
|
||||||
|
echo " --env '$(escape_for_shell "${BUILD_ENV[$i]}")'"
|
||||||
|
else
|
||||||
|
echo " --env '$(escape_for_shell "${BUILD_ENV[$i]}")' \\"
|
||||||
|
fi
|
||||||
|
done
|
||||||
echo ""
|
echo ""
|
||||||
echo "=========================================="
|
echo "=========================================="
|
||||||
echo -e "${YELLOW}To actually trigger this build, run:${NC}"
|
echo -e "${YELLOW}To actually trigger this build, run:${NC}"
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
REGISTRY="public.ecr.aws/q9t5s3a7"
|
||||||
|
REPO="vllm-release-repo"
|
||||||
|
ARCH_TAG="${BUILDKITE_COMMIT}-$(uname -m)-xpu"
|
||||||
|
PLATFORM_TAG="${BUILDKITE_COMMIT}-xpu"
|
||||||
|
|
||||||
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin ${REGISTRY}
|
||||||
|
docker manifest rm ${REGISTRY}/${REPO}:${PLATFORM_TAG} || true
|
||||||
|
docker manifest create ${REGISTRY}/${REPO}:${PLATFORM_TAG} ${REGISTRY}/${REPO}:${ARCH_TAG} --amend
|
||||||
|
docker manifest push ${REGISTRY}/${REPO}:${PLATFORM_TAG}
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
set -ex
|
||||||
|
|
||||||
|
ORIG_TAG_NAME="$BUILDKITE_COMMIT"
|
||||||
|
REPO="vllm/vllm-openai-xpu"
|
||||||
|
|
||||||
|
echo "Pushing original XPU tag ${ORIG_TAG_NAME}-xpu to nightly tags in ${REPO}"
|
||||||
|
|
||||||
|
aws ecr-public get-login-password --region us-east-1 | docker login --username AWS --password-stdin public.ecr.aws/q9t5s3a7
|
||||||
|
docker pull public.ecr.aws/q9t5s3a7/vllm-release-repo:"$ORIG_TAG_NAME"-x86_64-xpu
|
||||||
|
|
||||||
|
docker tag public.ecr.aws/q9t5s3a7/vllm-release-repo:"$ORIG_TAG_NAME"-x86_64-xpu ${REPO}:nightly-x86_64
|
||||||
|
docker push ${REPO}:nightly-x86_64
|
||||||
|
|
||||||
|
docker manifest rm ${REPO}:nightly || true
|
||||||
|
docker manifest rm ${REPO}:nightly-"$BUILDKITE_COMMIT" || true
|
||||||
|
docker manifest create ${REPO}:nightly ${REPO}:nightly-x86_64 --amend
|
||||||
|
docker manifest create ${REPO}:nightly-"$BUILDKITE_COMMIT" ${REPO}:nightly-x86_64 --amend
|
||||||
|
docker manifest push ${REPO}:nightly
|
||||||
|
docker manifest push ${REPO}:nightly-"$BUILDKITE_COMMIT"
|
||||||
+742
-641
File diff suppressed because it is too large
Load Diff
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: V1 attention (H100-MI300)
|
- label: V1 attention (H100-MI300)
|
||||||
key: v1-attention-h100-mi300
|
key: v1-attention-h100-mi300
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 85
|
||||||
device: h100
|
device: h100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/attention.py
|
- vllm/config/attention.py
|
||||||
@@ -12,11 +12,12 @@ steps:
|
|||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
- tests/v1/attention
|
- tests/v1/attention
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/attention
|
- pytest -v -s v1/attention --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 70
|
timeout_in_minutes: 95
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -30,7 +31,7 @@ steps:
|
|||||||
|
|
||||||
- label: V1 attention (B200)
|
- label: V1 attention (B200)
|
||||||
key: v1-attention-b200
|
key: v1-attention-b200
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 80
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/attention.py
|
- vllm/config/attention.py
|
||||||
@@ -38,4 +39,5 @@ steps:
|
|||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
- tests/v1/attention
|
- tests/v1/attention
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/attention
|
- pytest -v -s v1/attention --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
|
parallelism: 2
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Basic Correctness
|
- label: Basic Correctness
|
||||||
key: basic-correctness
|
key: basic-correctness
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -16,3 +16,9 @@ steps:
|
|||||||
- pytest -v -s basic_correctness/test_mem.py
|
- pytest -v -s basic_correctness/test_mem.py
|
||||||
- pytest -v -s basic_correctness/test_basic_correctness.py
|
- pytest -v -s basic_correctness/test_basic_correctness.py
|
||||||
- pytest -v -s basic_correctness/test_cpu_offload.py
|
- pytest -v -s basic_correctness/test_cpu_offload.py
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 70
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|||||||
@@ -4,13 +4,18 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Benchmarks CLI Test
|
- label: Benchmarks CLI Test
|
||||||
key: benchmarks-cli-test
|
key: benchmarks-cli-test
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 30
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/benchmarks/
|
- tests/benchmarks/
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s benchmarks/
|
- pytest -v -s benchmarks/
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: Attention Benchmarks Smoke Test (B200)
|
- label: Attention Benchmarks Smoke Test (B200)
|
||||||
key: attention-benchmarks-smoke-test-b200
|
key: attention-benchmarks-smoke-test-b200
|
||||||
@@ -18,9 +23,9 @@ steps:
|
|||||||
num_gpus: 2
|
num_gpus: 2
|
||||||
optional: true
|
optional: true
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 20
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- benchmarks/attention_benchmarks/
|
- benchmarks/attention_benchmarks/
|
||||||
- vllm/v1/attention/
|
- vllm/v1/attention/
|
||||||
commands:
|
commands:
|
||||||
- python3 benchmarks/attention_benchmarks/benchmark.py --backends flash flashinfer --batch-specs "8q1s1k" --repeats 1 --warmup-iters 1
|
- python3 benchmarks/attention_benchmarks/benchmark.py --backends flash flashinfer --batch-specs "8q1s1k"
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Sequence Parallel Correctness Tests (2 GPUs)
|
- label: Sequence Parallel Correctness Tests (2 GPUs)
|
||||||
key: sequence-parallel-correctness-tests-2-gpus
|
key: sequence-parallel-correctness-tests-2-gpus
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 80
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -19,7 +19,7 @@ steps:
|
|||||||
|
|
||||||
- label: Sequence Parallel Correctness Tests (2xH100)
|
- label: Sequence Parallel Correctness Tests (2xH100)
|
||||||
key: sequence-parallel-correctness-tests-2xh100
|
key: sequence-parallel-correctness-tests-2xh100
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 75
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
@@ -30,7 +30,7 @@ steps:
|
|||||||
|
|
||||||
- label: AsyncTP Correctness Tests (2xH100)
|
- label: AsyncTP Correctness Tests (2xH100)
|
||||||
key: asynctp-correctness-tests-2xh100
|
key: asynctp-correctness-tests-2xh100
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
@@ -41,7 +41,7 @@ steps:
|
|||||||
|
|
||||||
- label: AsyncTP Correctness Tests (B200)
|
- label: AsyncTP Correctness Tests (B200)
|
||||||
key: asynctp-correctness-tests-b200
|
key: asynctp-correctness-tests-b200
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
@@ -52,7 +52,7 @@ steps:
|
|||||||
|
|
||||||
- label: Distributed Compile Unit Tests (2xH100)
|
- label: Distributed Compile Unit Tests (2xH100)
|
||||||
key: distributed-compile-unit-tests-2xh100
|
key: distributed-compile-unit-tests-2xh100
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -66,7 +66,7 @@ steps:
|
|||||||
|
|
||||||
- label: Fusion and Compile Unit Tests (2xB200)
|
- label: Fusion and Compile Unit Tests (2xB200)
|
||||||
key: fusion-and-compile-unit-tests-2xb200
|
key: fusion-and-compile-unit-tests-2xb200
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -96,7 +96,7 @@ steps:
|
|||||||
|
|
||||||
- label: Fusion E2E Quick (H100)
|
- label: Fusion E2E Quick (H100)
|
||||||
key: fusion-e2e-quick-h100
|
key: fusion-e2e-quick-h100
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
@@ -115,7 +115,7 @@ steps:
|
|||||||
|
|
||||||
- label: Fusion E2E Config Sweep (H100)
|
- label: Fusion E2E Config Sweep (H100)
|
||||||
key: fusion-e2e-config-sweep-h100
|
key: fusion-e2e-config-sweep-h100
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
@@ -149,7 +149,7 @@ steps:
|
|||||||
|
|
||||||
- label: Fusion E2E TP2 Quick (H100)
|
- label: Fusion E2E TP2 Quick (H100)
|
||||||
key: fusion-e2e-tp2-quick-h100
|
key: fusion-e2e-tp2-quick-h100
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 35
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -167,7 +167,7 @@ steps:
|
|||||||
|
|
||||||
- label: Fusion E2E TP2 AR-RMS Config Sweep (H100)
|
- label: Fusion E2E TP2 AR-RMS Config Sweep (H100)
|
||||||
key: fusion-e2e-tp2-ar-rms-config-sweep-h100
|
key: fusion-e2e-tp2-ar-rms-config-sweep-h100
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -207,7 +207,7 @@ steps:
|
|||||||
|
|
||||||
- label: Fusion E2E TP2 (B200)
|
- label: Fusion E2E TP2 (B200)
|
||||||
key: fusion-e2e-tp2-b200
|
key: fusion-e2e-tp2-b200
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
|
|||||||
@@ -2,9 +2,9 @@ group: CUDA
|
|||||||
depends_on:
|
depends_on:
|
||||||
- image-build
|
- image-build
|
||||||
steps:
|
steps:
|
||||||
- label: Platform Tests (CUDA)
|
- label: Platform Tests
|
||||||
key: platform-tests-cuda
|
key: platform-tests
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 20
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/envs.py
|
- vllm/envs.py
|
||||||
@@ -19,7 +19,7 @@ steps:
|
|||||||
|
|
||||||
- label: Cudagraph
|
- label: Cudagraph
|
||||||
key: cudagraph
|
key: cudagraph
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 30
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- tests/v1/cudagraph
|
- tests/v1/cudagraph
|
||||||
- vllm/v1/cudagraph_dispatcher.py
|
- vllm/v1/cudagraph_dispatcher.py
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Distributed NixlConnector PD accuracy (4 GPUs)
|
- label: Distributed NixlConnector PD accuracy (4 GPUs)
|
||||||
key: distributed-nixlconnector-pd-accuracy-4-gpus
|
key: distributed-nixlconnector-pd-accuracy-4-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 55
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -13,9 +13,23 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_4
|
||||||
|
timeout_in_minutes: 85
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
commands:
|
||||||
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
|
- ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
- label: Distributed FlashInfer NixlConnector PD accuracy (4 GPUs)
|
- label: Distributed FlashInfer NixlConnector PD accuracy (4 GPUs)
|
||||||
key: distributed-flashinfer-nixlconnector-pd-accuracy-4-gpus
|
key: distributed-flashinfer-nixlconnector-pd-accuracy-4-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 55
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -25,6 +39,19 @@ steps:
|
|||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- FLASHINFER=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- FLASHINFER=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
|
- label: Push NixlConnector PP prefill PD accuracy (4 GPUs)
|
||||||
|
key: push-nixlconnector-pp-prefill-pd-accuracy-4-gpus
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
working_dir: "/vllm-workspace/tests"
|
||||||
|
num_devices: 4
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- tests/v1/kv_connector/nixl_push_integration/
|
||||||
|
commands:
|
||||||
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
|
- bash v1/kv_connector/nixl_push_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
- label: DP EP Distributed NixlConnector PD accuracy tests (4 GPUs)
|
- label: DP EP Distributed NixlConnector PD accuracy tests (4 GPUs)
|
||||||
key: dp-ep-distributed-nixlconnector-pd-accuracy-tests-4-gpus
|
key: dp-ep-distributed-nixlconnector-pd-accuracy-tests-4-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 30
|
||||||
@@ -36,10 +63,23 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- DP_EP=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- DP_EP=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_4
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
commands:
|
||||||
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
|
- DP_EP=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
- label: CrossLayer KV layout Distributed NixlConnector PD accuracy tests (4 GPUs)
|
- label: CrossLayer KV layout Distributed NixlConnector PD accuracy tests (4 GPUs)
|
||||||
key: crosslayer-kv-layout-distributed-nixlconnector-pd-accuracy-tests-4-gpus
|
key: crosslayer-kv-layout-distributed-nixlconnector-pd-accuracy-tests-4-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 55
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -48,10 +88,23 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- CROSS_LAYERS_BLOCKS=True bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- CROSS_LAYERS_BLOCKS=True bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_4
|
||||||
|
timeout_in_minutes: 85
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
commands:
|
||||||
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
|
- CROSS_LAYERS_BLOCKS=True ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
- label: Hybrid SSM NixlConnector PD accuracy tests (4 GPUs)
|
- label: Hybrid SSM NixlConnector PD accuracy tests (4 GPUs)
|
||||||
key: hybrid-ssm-nixlconnector-pd-accuracy-tests-4-gpus
|
key: hybrid-ssm-nixlconnector-pd-accuracy-tests-4-gpus
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 60
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -60,6 +113,19 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- HYBRID_SSM=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
- HYBRID_SSM=1 bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_4
|
||||||
|
timeout_in_minutes: 80
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
commands:
|
||||||
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
|
- HYBRID_SSM=1 ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_accuracy_test.sh
|
||||||
|
|
||||||
- label: Hybrid SSM NixlConnector PD prefix cache test (2 GPUs)
|
- label: Hybrid SSM NixlConnector PD prefix cache test (2 GPUs)
|
||||||
key: hybrid-ssm-nixlconnector-pd-prefix-cache-2-gpus
|
key: hybrid-ssm-nixlconnector-pd-prefix-cache-2-gpus
|
||||||
@@ -77,7 +143,7 @@ steps:
|
|||||||
|
|
||||||
- label: MultiConnector (Nixl+Offloading) PD accuracy (2 GPUs)
|
- label: MultiConnector (Nixl+Offloading) PD accuracy (2 GPUs)
|
||||||
key: multiconnector-nixl-offloading-pd-accuracy-2-gpus
|
key: multiconnector-nixl-offloading-pd-accuracy-2-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -92,7 +158,7 @@ steps:
|
|||||||
|
|
||||||
- label: NixlConnector PD + Spec Decode acceptance (2 GPUs)
|
- label: NixlConnector PD + Spec Decode acceptance (2 GPUs)
|
||||||
key: nixlconnector-pd-spec-decode-acceptance-2-gpus
|
key: nixlconnector-pd-spec-decode-acceptance-2-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: a100
|
device: a100
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -103,10 +169,24 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
- bash /vllm-workspace/.buildkite/scripts/install-kv-connectors.sh
|
||||||
- bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
- bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_2
|
||||||
|
timeout_in_minutes: 70
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/nixl/
|
||||||
|
- vllm/v1/worker/kv_connector_model_runner_mixin.py
|
||||||
|
- tests/v1/kv_connector/nixl_integration/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
commands:
|
||||||
|
- uv pip install --system -r /vllm-workspace/requirements/kv_connectors_rocm.txt
|
||||||
|
- ATTENTION_BACKEND=TRITON_ATTN bash v1/kv_connector/nixl_integration/config_sweep_spec_decode_test.sh
|
||||||
|
|
||||||
- label: MultiConnector (Nixl+Offloading) PD edge cases (2 GPUs)
|
- label: MultiConnector (Nixl+Offloading) PD edge cases (2 GPUs)
|
||||||
key: multiconnector-nixl-offloading-pd-edge-cases-2-gpus
|
key: multiconnector-nixl-offloading-pd-edge-cases-2-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Distributed Comm Ops
|
- label: Distributed Comm Ops
|
||||||
key: distributed-comm-ops
|
key: distributed-comm-ops
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -18,7 +18,7 @@ steps:
|
|||||||
|
|
||||||
- label: Distributed DP Tests (2 GPUs)
|
- label: Distributed DP Tests (2 GPUs)
|
||||||
key: distributed-dp-tests-2-gpus
|
key: distributed-dp-tests-2-gpus
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 35
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -37,10 +37,25 @@ steps:
|
|||||||
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_eagle_dp.py
|
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_eagle_dp.py
|
||||||
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_external_lb_dp.py
|
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_external_lb_dp.py
|
||||||
- DP_SIZE=2 pytest -v -s entrypoints/openai/test_multi_api_servers.py
|
- DP_SIZE=2 pytest -v -s entrypoints/openai/test_multi_api_servers.py
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_2
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/
|
||||||
|
- vllm/engine/
|
||||||
|
- vllm/executor/
|
||||||
|
- vllm/worker/worker_base.py
|
||||||
|
- vllm/v1/engine/
|
||||||
|
- vllm/v1/worker/
|
||||||
|
- tests/v1/distributed
|
||||||
|
- tests/entrypoints/openai/test_multi_api_servers.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: Distributed Compile + RPC Tests (2 GPUs)
|
- label: Distributed Compile + RPC Tests (2 GPUs)
|
||||||
key: distributed-compile-rpc-tests-2-gpus
|
key: distributed-compile-rpc-tests-2-gpus
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 65
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -63,7 +78,7 @@ steps:
|
|||||||
|
|
||||||
- label: Distributed Torchrun + Shutdown Tests (2 GPUs)
|
- label: Distributed Torchrun + Shutdown Tests (2 GPUs)
|
||||||
key: distributed-torchrun-shutdown-tests-2-gpus
|
key: distributed-torchrun-shutdown-tests-2-gpus
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -118,7 +133,7 @@ steps:
|
|||||||
|
|
||||||
- label: Distributed DP Tests (4 GPUs)
|
- label: Distributed DP Tests (4 GPUs)
|
||||||
key: distributed-dp-tests-4-gpus
|
key: distributed-dp-tests-4-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -139,7 +154,7 @@ steps:
|
|||||||
|
|
||||||
- label: Distributed Compile + Comm (4 GPUs)
|
- label: Distributed Compile + Comm (4 GPUs)
|
||||||
key: distributed-compile-comm-4-gpus
|
key: distributed-compile-comm-4-gpus
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 70
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -159,9 +174,9 @@ steps:
|
|||||||
# test multi-node TP with multiproc executor (simulated on single node)
|
# test multi-node TP with multiproc executor (simulated on single node)
|
||||||
- pytest -v -s distributed/test_multiproc_executor.py::test_multiproc_executor_multi_node
|
- pytest -v -s distributed/test_multiproc_executor.py::test_multiproc_executor_multi_node
|
||||||
|
|
||||||
- label: Distributed Tests (8 GPUs)(H100)
|
- label: Distributed Tests (8xH100)
|
||||||
key: distributed-tests-8-gpus-h100
|
key: distributed-tests-8xh100
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 20
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 8
|
num_devices: 8
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
@@ -180,8 +195,8 @@ steps:
|
|||||||
# test with torchrun tp=2 and dp=4 with ep
|
# test with torchrun tp=2 and dp=4 with ep
|
||||||
- torchrun --nproc-per-node=8 ../examples/features/torchrun/torchrun_dp_example_offline.py --tp-size=2 --pp-size=1 --dp-size=4 --enable-ep
|
- torchrun --nproc-per-node=8 ../examples/features/torchrun/torchrun_dp_example_offline.py --tp-size=2 --pp-size=1 --dp-size=4 --enable-ep
|
||||||
|
|
||||||
- label: Distributed Tests (4 GPUs)(A100)
|
- label: Distributed Tests (4xA100)
|
||||||
key: distributed-tests-4-gpus-a100
|
key: distributed-tests-4xa100
|
||||||
device: a100
|
device: a100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
@@ -195,9 +210,9 @@ steps:
|
|||||||
- TARGET_TEST_SUITE=A100 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
|
- TARGET_TEST_SUITE=A100 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
|
||||||
- pytest -v -s -x lora/test_mixtral.py
|
- pytest -v -s -x lora/test_mixtral.py
|
||||||
|
|
||||||
- label: Distributed Tests (2 GPUs)(H100)
|
- label: Distributed Tests (2xH100-2xMI300)
|
||||||
key: distributed-tests-2-gpus-h100
|
key: distributed-tests-2xh100-2xmi300
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 30
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
@@ -210,15 +225,15 @@ steps:
|
|||||||
- VLLM_ALLOW_INSECURE_SERIALIZATION=1 pytest -v -s tests/distributed/test_weight_transfer.py
|
- VLLM_ALLOW_INSECURE_SERIALIZATION=1 pytest -v -s tests/distributed/test_weight_transfer.py
|
||||||
- pytest -v -s tests/distributed/test_packed_tensor.py
|
- pytest -v -s tests/distributed/test_packed_tensor.py
|
||||||
|
|
||||||
- label: Distributed Tests (2 GPUs)(B200)
|
- label: Distributed Tests (2xB200)
|
||||||
key: distributed-tests-2-gpus-b200
|
key: distributed-tests-2xb200
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s tests/distributed/test_context_parallel.py
|
- pytest -v -s tests/distributed/test_context_parallel.py
|
||||||
- pytest -v -s tests/distributed/test_nccl_symm_mem_allreduce.py
|
- pytest -v -s tests/distributed/test_nccl_symm_mem.py
|
||||||
- pytest -v -s tests/v1/distributed/test_dbo.py
|
- pytest -v -s tests/v1/distributed/test_dbo.py
|
||||||
- pytest -v -s tests/distributed/test_mnnvl_alltoall.py
|
- pytest -v -s tests/distributed/test_mnnvl_alltoall.py
|
||||||
|
|
||||||
@@ -244,7 +259,7 @@ steps:
|
|||||||
|
|
||||||
- label: Pipeline + Context Parallelism (4 GPUs)
|
- label: Pipeline + Context Parallelism (4 GPUs)
|
||||||
key: pipeline-context-parallelism-4-gpus
|
key: pipeline-context-parallelism-4-gpus
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 55
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -259,7 +274,7 @@ steps:
|
|||||||
|
|
||||||
- label: RayExecutorV2 (4 GPUs)
|
- label: RayExecutorV2 (4 GPUs)
|
||||||
key: rayexecutorv2-4-gpus
|
key: rayexecutorv2-4-gpus
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ depends_on:
|
|||||||
- image-build-cpu
|
- image-build-cpu
|
||||||
steps:
|
steps:
|
||||||
- label: Docker Build Metadata
|
- label: Docker Build Metadata
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 20
|
||||||
device: cpu-small
|
device: cpu-small
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- .buildkite/release-pipeline.yaml
|
- .buildkite/release-pipeline.yaml
|
||||||
|
|||||||
@@ -2,9 +2,9 @@ group: E2E Integration
|
|||||||
depends_on:
|
depends_on:
|
||||||
- image-build
|
- image-build
|
||||||
steps:
|
steps:
|
||||||
- label: DeepSeek V2-Lite Sync EPLB Accuracy
|
- label: DeepSeek V2-Lite Sync EPLB Accuracy (4xH100)
|
||||||
key: deepseek-v2-lite-sync-eplb-accuracy
|
key: deepseek-v2-lite-sync-eplb-accuracy-4xh100
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 25
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
@@ -12,9 +12,9 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash .buildkite/scripts/scheduled_integration_test/deepseek_v2_lite_ep_eplb.sh 0.25 200 8010
|
- bash .buildkite/scripts/scheduled_integration_test/deepseek_v2_lite_ep_eplb.sh 0.25 200 8010
|
||||||
|
|
||||||
- label: Qwen3-30B-A3B-FP8-block Sync EPLB Accuracy
|
- label: Qwen3-30B-A3B-FP8-block Sync EPLB Accuracy (4xH100)
|
||||||
key: qwen3-30b-a3b-fp8-block-sync-eplb-accuracy
|
key: qwen3-30b-a3b-fp8-block-sync-eplb-accuracy-4xh100
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 25
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
@@ -22,9 +22,9 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- bash .buildkite/scripts/scheduled_integration_test/qwen30b_a3b_fp8_block_ep_eplb.sh 0.8 200 8020
|
- bash .buildkite/scripts/scheduled_integration_test/qwen30b_a3b_fp8_block_ep_eplb.sh 0.8 200 8020
|
||||||
|
|
||||||
- label: Qwen3-30B-A3B-FP8-block Sync EPLB Accuracy (B200)
|
- label: Qwen3-30B-A3B-FP8-block Sync EPLB Accuracy (2xB200)
|
||||||
key: qwen3-30b-a3b-fp8-block-sync-eplb-accuracy-b200
|
key: qwen3-30b-a3b-fp8-block-sync-eplb-accuracy-2xb200
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 20
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -34,7 +34,7 @@ steps:
|
|||||||
|
|
||||||
- label: Qwen3-30B-A3B-FP8 DP4 Async EPLB Accuracy
|
- label: Qwen3-30B-A3B-FP8 DP4 Async EPLB Accuracy
|
||||||
key: qwen3-30b-a3b-fp8-dp4-async-eplb-accuracy
|
key: qwen3-30b-a3b-fp8-dp4-async-eplb-accuracy
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 25
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
@@ -44,7 +44,7 @@ steps:
|
|||||||
|
|
||||||
- label: DeepSeek V2-Lite Prefetch Offload Accuracy (H100)
|
- label: DeepSeek V2-Lite Prefetch Offload Accuracy (H100)
|
||||||
key: deepseek-v2-lite-prefetch-offload-accuracy-h100
|
key: deepseek-v2-lite-prefetch-offload-accuracy-h100
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 20
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Engine
|
- label: Engine
|
||||||
key: engine
|
key: engine
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 30
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/compilation/
|
- vllm/compilation/
|
||||||
@@ -29,13 +29,13 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 50
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Engine (1 GPU)
|
- label: Engine (1 GPU)
|
||||||
key: engine-1-gpu
|
key: engine-1-gpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/engine/
|
- vllm/v1/engine/
|
||||||
- tests/v1/engine/
|
- tests/v1/engine/
|
||||||
@@ -45,13 +45,13 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: e2e Scheduling (1 GPU)
|
- label: e2e Scheduling (1 GPU)
|
||||||
key: e2e-scheduling-1-gpu
|
key: e2e-scheduling-1-gpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 35
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/
|
- vllm/v1/
|
||||||
@@ -60,24 +60,34 @@ steps:
|
|||||||
- pytest -v -s v1/e2e/general/test_async_scheduling.py
|
- pytest -v -s v1/e2e/general/test_async_scheduling.py
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi250_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 70
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: e2e Core (1 GPU)
|
- label: e2e Core (1 GPU)
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: e2e-core-1-gpu
|
key: e2e-core-1-gpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/
|
- vllm/v1/
|
||||||
- tests/v1/e2e/general/
|
- tests/v1/e2e/general/
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py
|
- pytest -v -s v1/e2e/general --ignore v1/e2e/general/test_async_scheduling.py
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/
|
||||||
|
- tests/v1/e2e/general/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: V1 e2e (2 GPUs)
|
- label: V1 e2e (2 GPUs)
|
||||||
key: v1-e2e-2-gpus
|
key: v1-e2e-2-gpus
|
||||||
timeout_in_minutes: 60 # TODO: Fix timeout after we have more confidence in the test stability
|
timeout_in_minutes: 25 # TODO: Fix timeout after we have more confidence in the test stability
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -102,10 +112,15 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
# Only run tests that need exactly 2 GPUs
|
# Only run tests that need exactly 2 GPUs
|
||||||
- pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "tensor_parallelism"
|
- pytest -v -s v1/e2e/spec_decode/test_spec_decode.py -k "tensor_parallelism"
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_2
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: V1 e2e (4 GPUs)
|
- label: V1 e2e (4 GPUs)
|
||||||
key: v1-e2e-4-gpus
|
key: v1-e2e-4-gpus
|
||||||
timeout_in_minutes: 60 # TODO: Fix timeout after we have more confidence in the test stability
|
timeout_in_minutes: 20 # TODO: Fix timeout after we have more confidence in the test stability
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -133,7 +148,7 @@ steps:
|
|||||||
|
|
||||||
- label: V1 e2e (4xH100)
|
- label: V1 e2e (4xH100)
|
||||||
key: v1-e2e-4xh100
|
key: v1-e2e-4xh100
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 35
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
optional: true
|
optional: true
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Entrypoints Unit Tests
|
- label: Entrypoints Unit Tests
|
||||||
key: entrypoints-unit-tests
|
key: entrypoints-unit-tests
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/entrypoints
|
- vllm/entrypoints
|
||||||
@@ -16,7 +16,7 @@ steps:
|
|||||||
|
|
||||||
- label: Entrypoints Integration (LLM)
|
- label: Entrypoints Integration (LLM)
|
||||||
key: entrypoints-integration-llm
|
key: entrypoints-integration-llm
|
||||||
timeout_in_minutes: 40
|
timeout_in_minutes: 60
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -29,21 +29,25 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
|
# TODO(akaratza): Test after Torch >= 2.12 bump
|
||||||
|
soft_fail: true
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (API Server)
|
- label: Entrypoints Integration (API Server)
|
||||||
key: entrypoints-integration-api-server
|
key: entrypoints-integration-api-server
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
timeout_in_minutes: 130
|
timeout_in_minutes: 50
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/entrypoints/serve
|
- tests/entrypoints/serve
|
||||||
|
- tests/entrypoints/scale_out
|
||||||
commands:
|
commands:
|
||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
- pytest -v -s entrypoints/serve --ignore=entrypoints/serve/dev/rpc
|
- pytest -v -s entrypoints/serve --ignore=entrypoints/serve/dev/rpc
|
||||||
- PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc
|
- PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc
|
||||||
|
- pytest -v -s entrypoints/scale_out
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
@@ -52,7 +56,7 @@ steps:
|
|||||||
|
|
||||||
- label: Entrypoints Integration (API Server OpenAI - Part 1)
|
- label: Entrypoints Integration (API Server OpenAI - Part 1)
|
||||||
key: entrypoints-integration-api-server-openai-part-1
|
key: entrypoints-integration-api-server-openai-part-1
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -64,13 +68,13 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 80
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
- label: Entrypoints Integration (API Server OpenAI - Part 2)
|
- label: Entrypoints Integration (API Server OpenAI - Part 2)
|
||||||
key: entrypoints-integration-api-server-openai-part-2
|
key: entrypoints-integration-api-server-openai-part-2
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -105,7 +109,7 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 65
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
|
|
||||||
@@ -122,7 +126,7 @@ steps:
|
|||||||
- label: Entrypoints Integration (Speech to Text)
|
- label: Entrypoints Integration (Speech to Text)
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: entrypoints-integration-speech_to_text
|
key: entrypoints-integration-speech_to_text
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -134,7 +138,7 @@ steps:
|
|||||||
- label: Entrypoints Integration (Multimodal)
|
- label: Entrypoints Integration (Multimodal)
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: entrypoints-integration-multimodal
|
key: entrypoints-integration-multimodal
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 45
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -156,7 +160,7 @@ steps:
|
|||||||
|
|
||||||
- label: OpenAI API Correctness
|
- label: OpenAI API Correctness
|
||||||
key: openai-api-correctness
|
key: openai-api-correctness
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/
|
- csrc/
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: EPLB Algorithm
|
- label: EPLB Algorithm
|
||||||
key: eplb-algorithm
|
key: eplb-algorithm
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 20
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -14,10 +14,20 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -v -s distributed/test_eplb_algo.py
|
- pytest -v -s distributed/test_eplb_algo.py
|
||||||
- pytest -v -s distributed/test_eplb_utils.py
|
- pytest -v -s distributed/test_eplb_utils.py
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/eplb
|
||||||
|
- tests/distributed/test_eplb_algo.py
|
||||||
|
- tests/distributed/test_eplb_utils.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: EPLB Execution # 17min
|
- label: EPLB Execution # 17min
|
||||||
key: eplb-execution
|
key: eplb-execution
|
||||||
timeout_in_minutes: 27
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -29,7 +39,7 @@ steps:
|
|||||||
|
|
||||||
- label: Elastic EP Scaling Test
|
- label: Elastic EP Scaling Test
|
||||||
key: elastic-ep-scaling-test
|
key: elastic-ep-scaling-test
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 30
|
||||||
device: h100
|
device: h100
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: vLLM IR Tests
|
- label: vLLM IR Tests
|
||||||
key: vllm-ir-tests
|
key: vllm-ir-tests
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 35
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -16,18 +16,19 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels Core Operation Test
|
- label: Kernels Core Operation Test
|
||||||
key: kernels-core-operation-test
|
key: kernels-core-operation-test
|
||||||
timeout_in_minutes: 75
|
timeout_in_minutes: 120
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/
|
- csrc/
|
||||||
- tests/kernels/core
|
- tests/kernels/core
|
||||||
- tests/kernels/test_concat_mla_q.py
|
- tests/kernels/test_concat_mla_q.py
|
||||||
- tests/kernels/test_fused_qk_norm_rope_gate.py
|
- tests/kernels/test_fused_qk_norm_rope_gate.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s kernels/core --ignore=kernels/core/test_minimax_reduce_rms.py kernels/test_concat_mla_q.py kernels/test_fused_qk_norm_rope_gate.py
|
- pytest -v -s kernels/core --ignore=kernels/core/test_minimax_reduce_rms.py kernels/test_concat_mla_q.py kernels/test_fused_qk_norm_rope_gate.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
|
parallelism: 3
|
||||||
|
|
||||||
- label: Kernels MiniMax Reduce RMS Test (2 GPUs)
|
- label: Kernels MiniMax Reduce RMS Test (2 GPUs)
|
||||||
key: kernels-minimax-reduce-rms-test-2-gpus
|
key: kernels-minimax-reduce-rms-test-2-gpus
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 20
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
device: h100
|
device: h100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -41,18 +42,20 @@ steps:
|
|||||||
|
|
||||||
- label: Deepseek V4 Kernel Test (H100)
|
- label: Deepseek V4 Kernel Test (H100)
|
||||||
key: deepseek-v4-kernel-test-h100
|
key: deepseek-v4-kernel-test-h100
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 30
|
||||||
device: h100
|
device: h100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
|
- csrc/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
|
||||||
- vllm/models/deepseek_v4/common/ops/
|
- vllm/models/deepseek_v4/common/ops/
|
||||||
- tests/kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py
|
- tests/kernels/test_fused_deepseek_v4_qnorm_rope_kv_insert.py
|
||||||
|
- tests/kernels/test_top_k_per_row.py # it runs on Blackwell too - some kernels have arch-specific optimizations
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s kernels/test_fused_deepseek_v4_*.py
|
- pytest -v -s kernels/test_fused_deepseek_v4_*.py
|
||||||
|
- pytest -v -s kernels/test_top_k_per_row.py
|
||||||
|
|
||||||
- label: Deepseek V4 Kernel Test (B200)
|
- label: Deepseek V4 Kernel Test (B200)
|
||||||
key: deepseek-v4-kernel-test-b200
|
key: deepseek-v4-kernel-test-b200
|
||||||
timeout_in_minutes: 15
|
timeout_in_minutes: 20
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
|
- csrc/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
|
||||||
@@ -63,7 +66,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels Attention Test %N
|
- label: Kernels Attention Test %N
|
||||||
key: kernels-attention-test
|
key: kernels-attention-test
|
||||||
timeout_in_minutes: 35
|
timeout_in_minutes: 65
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/attention/
|
- csrc/attention/
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -74,6 +77,20 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -v -s kernels/attention --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
- pytest -v -s kernels/attention --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
parallelism: 2
|
parallelism: 2
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 90
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/attention/
|
||||||
|
- vllm/v1/attention
|
||||||
|
- vllm/model_executor/layers/attention
|
||||||
|
- tests/kernels/attention
|
||||||
|
- vllm/_aiter_ops.py
|
||||||
|
- vllm/envs.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: Kernels Attention DiffKV Test (H100)
|
- label: Kernels Attention DiffKV Test (H100)
|
||||||
key: kernels-attention-diffkv-test-h100
|
key: kernels-attention-diffkv-test-h100
|
||||||
@@ -90,7 +107,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels Quantization Test %N
|
- label: Kernels Quantization Test %N
|
||||||
key: kernels-quantization-test
|
key: kernels-quantization-test
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 60
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/quantization/
|
- csrc/quantization/
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
@@ -104,6 +121,7 @@ steps:
|
|||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/quantization/
|
- csrc/quantization/
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
|
- vllm/config/
|
||||||
- tests/kernels/quantization
|
- tests/kernels/quantization
|
||||||
- tests/kernels/quantization/test_rocm_skinny_gemms.py
|
- tests/kernels/quantization/test_rocm_skinny_gemms.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
@@ -114,7 +132,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels MoE Test %N
|
- label: Kernels MoE Test %N
|
||||||
key: kernels-moe-test
|
key: kernels-moe-test
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 50
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/quantization/cutlass_w8a8/moe/
|
- csrc/quantization/cutlass_w8a8/moe/
|
||||||
- csrc/moe/
|
- csrc/moe/
|
||||||
@@ -127,10 +145,26 @@ steps:
|
|||||||
- pytest -v -s kernels/moe --ignore=kernels/moe/test_modular_oai_triton_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
- pytest -v -s kernels/moe --ignore=kernels/moe/test_modular_oai_triton_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
- pytest -v -s kernels/moe/test_modular_oai_triton_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
- pytest -v -s kernels/moe/test_modular_oai_triton_moe.py --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
parallelism: 5
|
parallelism: 5
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 65
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/quantization/cutlass_w8a8/moe/
|
||||||
|
- csrc/moe/
|
||||||
|
- tests/kernels/moe
|
||||||
|
- vllm/model_executor/layers/fused_moe/
|
||||||
|
- vllm/distributed/device_communicators/
|
||||||
|
- vllm/envs.py
|
||||||
|
- vllm/config
|
||||||
|
- vllm/_aiter_ops.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: Kernels Mamba Test
|
- label: Kernels Mamba Test
|
||||||
key: kernels-mamba-test
|
key: kernels-mamba-test
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/mamba/
|
- csrc/mamba/
|
||||||
- tests/kernels/mamba
|
- tests/kernels/mamba
|
||||||
@@ -139,7 +173,7 @@ steps:
|
|||||||
- pytest -v -s kernels/mamba
|
- pytest -v -s kernels/mamba
|
||||||
|
|
||||||
- label: Kernels KDA Test
|
- label: Kernels KDA Test
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 25
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/layers/fla/ops/kda.py
|
- vllm/model_executor/layers/fla/ops/kda.py
|
||||||
@@ -151,7 +185,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels DeepGEMM Test (H100)
|
- label: Kernels DeepGEMM Test (H100)
|
||||||
key: kernels-deepgemm-test-h100
|
key: kernels-deepgemm-test-h100
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 35
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -178,7 +212,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels (B200)
|
- label: Kernels (B200)
|
||||||
key: kernels-b200
|
key: kernels-b200
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 80
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
# optional: true
|
# optional: true
|
||||||
@@ -231,19 +265,20 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels Helion Test
|
- label: Kernels Helion Test
|
||||||
key: kernels-helion-test
|
key: kernels-helion-test
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 115
|
||||||
device: h100
|
device: h100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/utils/import_utils.py
|
- vllm/utils/import_utils.py
|
||||||
- tests/kernels/helion/
|
- tests/kernels/helion/
|
||||||
commands:
|
commands:
|
||||||
- pip install helion==1.1.0
|
- pip install helion==1.1.0
|
||||||
- pytest -v -s kernels/helion/
|
- pytest -v -s kernels/helion/ --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT
|
||||||
|
parallelism: 2
|
||||||
|
|
||||||
|
|
||||||
- label: Kernels FP8 MoE Test (1 H100)
|
- label: Kernels FP8 MoE Test (1xH100)
|
||||||
key: kernels-fp8-moe-test-1-h100
|
key: kernels-fp8-moe-test-1xh100
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 40
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
optional: true
|
optional: true
|
||||||
@@ -258,9 +293,9 @@ steps:
|
|||||||
- pytest -v -s kernels/moe/test_triton_moe_no_act_mul.py
|
- pytest -v -s kernels/moe/test_triton_moe_no_act_mul.py
|
||||||
- pytest -v -s kernels/moe/test_triton_moe_ptpc_fp8.py
|
- pytest -v -s kernels/moe/test_triton_moe_ptpc_fp8.py
|
||||||
|
|
||||||
- label: Kernels FP8 MoE Test (2 H100s)
|
- label: Kernels FP8 MoE Test (2xH100)
|
||||||
key: kernels-fp8-moe-test-2-h100s
|
key: kernels-fp8-moe-test-2xh100
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 45
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
optional: true
|
optional: true
|
||||||
@@ -270,7 +305,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels Fp4 MoE Test (B200)
|
- label: Kernels Fp4 MoE Test (B200)
|
||||||
key: kernels-fp4-moe-test-b200
|
key: kernels-fp4-moe-test-b200
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 25
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
optional: true
|
optional: true
|
||||||
@@ -283,7 +318,7 @@ steps:
|
|||||||
|
|
||||||
- label: Kernels FusedMoE Layer Test (2 H100s)
|
- label: Kernels FusedMoE Layer Test (2 H100s)
|
||||||
key: kernels-fusedmoe-layer-test-2-h100s
|
key: kernels-fusedmoe-layer-test-2-h100s
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 30
|
||||||
device: h100
|
device: h100
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ steps:
|
|||||||
- label: LM Eval Small Models
|
- label: LM Eval Small Models
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: lm-eval-small-models
|
key: lm-eval-small-models
|
||||||
timeout_in_minutes: 75
|
timeout_in_minutes: 45
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/
|
- csrc/
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
@@ -28,7 +28,8 @@ steps:
|
|||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
- vllm/platforms/rocm.py
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
# - label: LM Eval Large Models (4 GPUs)(A100)
|
# - label: LM Eval Large Models (4xA100)
|
||||||
|
# key: lm-eval-large-models-4xa100
|
||||||
# device: a100
|
# device: a100
|
||||||
# optional: true
|
# optional: true
|
||||||
# num_devices: 4
|
# num_devices: 4
|
||||||
@@ -40,8 +41,8 @@ steps:
|
|||||||
# - export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
# - export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
# - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large.txt --tp-size=4
|
# - pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large.txt --tp-size=4
|
||||||
|
|
||||||
- label: LM Eval Large Models (4 GPUs)(H100)
|
- label: LM Eval Large Models (4xH100)
|
||||||
key: lm-eval-large-models-4-gpus-h100
|
key: lm-eval-large-models-4xh100
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
@@ -53,9 +54,9 @@ steps:
|
|||||||
- export VLLM_USE_DEEP_GEMM=0 # We found Triton is faster than DeepGEMM for H100
|
- export VLLM_USE_DEEP_GEMM=0 # We found Triton is faster than DeepGEMM for H100
|
||||||
- pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-hopper.txt --tp-size=4
|
- pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-large-hopper.txt --tp-size=4
|
||||||
|
|
||||||
- label: LM Eval Small Models (B200)
|
- label: LM Eval Small Models (1xB200)
|
||||||
key: lm-eval-small-models-b200
|
key: lm-eval-small-models-1xb200
|
||||||
timeout_in_minutes: 120
|
timeout_in_minutes: 50
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -64,10 +65,23 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell.txt
|
||||||
|
|
||||||
- label: LM Eval Large Models (B200, EP)
|
- label: LM Eval Small Models Distributed (2xB200)
|
||||||
key: lm-eval-large-models-b200-ep
|
key: lm-eval-small-models-distributed-2xb200
|
||||||
timeout_in_minutes: 120
|
timeout_in_minutes: 120
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
|
num_devices: 2
|
||||||
|
optional: true
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/
|
||||||
|
- vllm/model_executor/layers/quantization
|
||||||
|
autorun_on_main: true
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-small-tp.txt
|
||||||
|
|
||||||
|
- label: LM Eval Large Models EP (2xB200)
|
||||||
|
key: lm-eval-large-models-ep-2xb200
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -76,9 +90,9 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell-ep.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-blackwell-ep.txt
|
||||||
|
|
||||||
- label: LM Eval Qwen3.5 Models (B200)
|
- label: LM Eval Qwen3.5 Models (2xB200)
|
||||||
key: lm-eval-qwen3-5-models-b200
|
key: lm-eval-qwen3-5-models-2xb200
|
||||||
timeout_in_minutes: 120
|
timeout_in_minutes: 45
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -93,14 +107,24 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-blackwell.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-qwen35-blackwell.txt
|
||||||
|
|
||||||
- label: LM Eval Large Models (H200)
|
- label: LM Eval Large Models (8xH200)
|
||||||
key: lm-eval-large-models-h200
|
key: lm-eval-large-models-8xh200
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 50
|
||||||
device: h200
|
device: h200
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 8
|
num_devices: 8
|
||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-h200.txt
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_8
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
commands:
|
||||||
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
|
- export PYTORCH_ROCM_ARCH=gfx942 # Limit Quark compilation to save time
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=configs/models-mi3xx.txt
|
||||||
|
|
||||||
- label: MoE Refactor Integration Test (H100 - TEMPORARY)
|
- label: MoE Refactor Integration Test (H100 - TEMPORARY)
|
||||||
key: moe-refactor-integration-test-h100-temporary
|
key: moe-refactor-integration-test-h100-temporary
|
||||||
@@ -126,10 +150,101 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor-dp-ep/config-b200.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/moe-refactor-dp-ep/config-b200.txt
|
||||||
|
|
||||||
|
- label: LM Eval Humming f16 (A100 - TEMPORARY)
|
||||||
|
key: lm-eval-humming-f16-a100
|
||||||
|
timeout_in_minutes: 75
|
||||||
|
device: a100
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers/quantization/humming.py
|
||||||
|
- vllm/model_executor/layers/quantization/utils/humming_utils.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/oracle/
|
||||||
|
- vllm/model_executor/kernels/linear/
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config.txt
|
||||||
|
|
||||||
|
- label: LM Eval Humming Act int8 (A100 - TEMPORARY)
|
||||||
|
key: lm-eval-humming-act-a100
|
||||||
|
timeout_in_minutes: 45
|
||||||
|
device: a100
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers/quantization/humming.py
|
||||||
|
- vllm/model_executor/layers/quantization/utils/humming_utils.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/oracle/
|
||||||
|
- vllm/model_executor/kernels/linear/
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-int8.txt
|
||||||
|
|
||||||
|
- label: LM Eval Humming f16 (H100 - TEMPORARY)
|
||||||
|
key: lm-eval-humming-f16-h100
|
||||||
|
timeout_in_minutes: 70
|
||||||
|
device: h100
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers/quantization/humming.py
|
||||||
|
- vllm/model_executor/layers/quantization/utils/humming_utils.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/oracle/
|
||||||
|
- vllm/model_executor/kernels/linear/
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config.txt
|
||||||
|
|
||||||
|
- label: LM Eval Humming Act fp8/int8 (H100 - TEMPORARY)
|
||||||
|
key: lm-eval-humming-act-h100
|
||||||
|
timeout_in_minutes: 70
|
||||||
|
device: h100
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers/quantization/humming.py
|
||||||
|
- vllm/model_executor/layers/quantization/utils/humming_utils.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/oracle/
|
||||||
|
- vllm/model_executor/kernels/linear/
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-fp8.txt
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-int8.txt
|
||||||
|
|
||||||
|
- label: LM Eval Humming f16 (B200 - TEMPORARY)
|
||||||
|
key: lm-eval-humming-f16-b200
|
||||||
|
timeout_in_minutes: 50
|
||||||
|
device: b200-k8s
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers/quantization/humming.py
|
||||||
|
- vllm/model_executor/layers/quantization/utils/humming_utils.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/oracle/
|
||||||
|
- vllm/model_executor/kernels/linear/
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config.txt
|
||||||
|
|
||||||
|
- label: LM Eval Humming Act fp8/int8 (B200 - TEMPORARY)
|
||||||
|
key: lm-eval-humming-act-b200
|
||||||
|
timeout_in_minutes: 50
|
||||||
|
device: b200-k8s
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/layers/quantization/humming.py
|
||||||
|
- vllm/model_executor/layers/quantization/utils/humming_utils.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/experts/fused_humming_moe.py
|
||||||
|
- vllm/model_executor/layers/fused_moe/oracle/
|
||||||
|
- vllm/model_executor/kernels/linear/
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-fp8.txt
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/humming/config-act-int8.txt
|
||||||
|
|
||||||
- label: LM Eval TurboQuant KV Cache
|
- label: LM Eval TurboQuant KV Cache
|
||||||
key: lm-eval-turboquant-kv-cache
|
key: lm-eval-turboquant-kv-cache
|
||||||
timeout_in_minutes: 75
|
timeout_in_minutes: 55
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/layers/quantization/turboquant/
|
- vllm/model_executor/layers/quantization/turboquant/
|
||||||
@@ -139,9 +254,9 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant.txt
|
- pytest -s -v evals/gsm8k/test_gsm8k_correctness.py --config-list-file=evals/gsm8k/configs/models-turboquant.txt
|
||||||
|
|
||||||
- label: GPQA Eval (GPT-OSS) (H100)
|
- label: GPQA Eval (GPT-OSS) (2xH100)
|
||||||
key: gpqa-eval-gpt-oss-h100
|
key: gpqa-eval-gpt-oss-2xh100
|
||||||
timeout_in_minutes: 120
|
timeout_in_minutes: 35
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -153,9 +268,9 @@ steps:
|
|||||||
- uv pip install --system 'gpt-oss[eval]==0.0.5'
|
- uv pip install --system 'gpt-oss[eval]==0.0.5'
|
||||||
- pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-h100.txt
|
- pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-h100.txt
|
||||||
|
|
||||||
- label: GPQA Eval (GPT-OSS) (B200)
|
- label: GPQA Eval (GPT-OSS) (2xB200)
|
||||||
key: gpqa-eval-gpt-oss-b200
|
key: gpqa-eval-gpt-oss-2xb200
|
||||||
timeout_in_minutes: 120
|
timeout_in_minutes: 30
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
@@ -167,9 +282,66 @@ steps:
|
|||||||
- uv pip install --system 'gpt-oss[eval]==0.0.5'
|
- uv pip install --system 'gpt-oss[eval]==0.0.5'
|
||||||
- pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-b200.txt
|
- pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-b200.txt
|
||||||
|
|
||||||
|
- label: GPQA Eval (GPT-OSS) (DGX Spark)
|
||||||
|
key: gpqa-eval-gpt-oss-spark
|
||||||
|
timeout_in_minutes: 35
|
||||||
|
device: dgx-spark
|
||||||
|
optional: true
|
||||||
|
num_devices: 1
|
||||||
|
depends_on:
|
||||||
|
- arm64-image-build
|
||||||
|
source_file_dependencies:
|
||||||
|
- csrc/
|
||||||
|
- vllm/model_executor/layers/quantization
|
||||||
|
- tests/evals/gpt_oss/
|
||||||
|
commands:
|
||||||
|
- uv pip install --system 'gpt-oss[eval]==0.0.5'
|
||||||
|
- pytest -s -v evals/gpt_oss/test_gpqa_correctness.py --config-list-file=configs/models-spark.txt
|
||||||
|
|
||||||
|
- label: LM Eval KV-Offload (1xH200)
|
||||||
|
key: kv-offload-small
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
device: h200_35gb
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/offloading/
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
|
||||||
|
- vllm/v1/kv_offload/
|
||||||
|
- vllm/v1/simple_kv_offload/
|
||||||
|
- tests/evals/gsm8k/test_gsm8k_offloading.py
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "nemotron-h-8b or gemma-4-e4b-it"
|
||||||
|
|
||||||
|
- label: LM Eval KV-Offload (2xH100)
|
||||||
|
key: kv-offload-medium
|
||||||
|
timeout_in_minutes: 30
|
||||||
|
device: h100
|
||||||
|
num_devices: 2
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/offloading/
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
|
||||||
|
- vllm/v1/kv_offload/
|
||||||
|
- vllm/v1/simple_kv_offload/
|
||||||
|
- tests/evals/gsm8k/test_gsm8k_offloading.py
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "qwen3.5-35b"
|
||||||
|
|
||||||
|
- label: LM Eval KV-Offload (4xH100)
|
||||||
|
key: kv-offload-large
|
||||||
|
timeout_in_minutes: 40
|
||||||
|
device: h100
|
||||||
|
num_devices: 4
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/offloading/
|
||||||
|
- vllm/distributed/kv_transfer/kv_connector/v1/simple_cpu_offload_connector.py
|
||||||
|
- vllm/v1/kv_offload/
|
||||||
|
- vllm/v1/simple_kv_offload/
|
||||||
|
- tests/evals/gsm8k/test_gsm8k_offloading.py
|
||||||
|
commands:
|
||||||
|
- pytest -s -v evals/gsm8k/test_gsm8k_offloading.py -k "deepseek-v4-flash"
|
||||||
|
|
||||||
- label: MRCR Eval Small Models
|
- label: MRCR Eval Small Models
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 25
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- tests/evals/mrcr/
|
- tests/evals/mrcr/
|
||||||
commands:
|
commands:
|
||||||
|
|||||||
@@ -5,18 +5,29 @@ steps:
|
|||||||
- label: LoRA %N
|
- label: LoRA %N
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: lora
|
key: lora
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/lora
|
- vllm/lora
|
||||||
- tests/lora
|
- tests/lora
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_qwen3_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py
|
- pytest -v -s lora --shard-id=$$BUILDKITE_PARALLEL_JOB --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --ignore=lora/test_chatglm3_tp.py --ignore=lora/test_llama_tp.py --ignore=lora/test_qwen3_with_multi_loras.py --ignore=lora/test_olmoe_tp.py --ignore=lora/test_deepseekv2_tp.py --ignore=lora/test_gptoss_tp.py --ignore=lora/test_qwen3moe_tp.py --ignore=lora/test_qwen35_densemodel_lora.py
|
||||||
parallelism: 4
|
parallelism: 4
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
working_dir: "/vllm-workspace/tests"
|
||||||
|
timeout_in_minutes: 65
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/lora
|
||||||
|
- tests/lora
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
|
|
||||||
- label: LoRA TP (Distributed)
|
- label: LoRA TP (Distributed)
|
||||||
key: lora-tp-distributed
|
key: lora-tp-distributed
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 60
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/lora
|
- vllm/lora
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ steps:
|
|||||||
- label: V1 Spec Decode
|
- label: V1 Spec Decode
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: v1-spec-decode
|
key: v1-spec-decode
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
- vllm/distributed/
|
- vllm/distributed/
|
||||||
@@ -21,10 +21,16 @@ steps:
|
|||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
# TODO: create another `optional` test group for slow tests
|
# TODO: create another `optional` test group for slow tests
|
||||||
- pytest -v -s -m 'not slow_test' v1/spec_decode
|
- pytest -v -s -m 'not slow_test' v1/spec_decode
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 75
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: V1 Sample + Logits
|
- label: V1 Sample + Logits
|
||||||
key: v1-sample-logits
|
key: v1-sample-logits
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
@@ -58,7 +64,7 @@ steps:
|
|||||||
|
|
||||||
- label: V1 Core + KV + Metrics
|
- label: V1 Core + KV + Metrics
|
||||||
key: v1-core-kv-metrics
|
key: v1-core-kv-metrics
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 60
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
- vllm/distributed/
|
- vllm/distributed/
|
||||||
@@ -83,6 +89,7 @@ steps:
|
|||||||
- tests/v1/simple_kv_offload
|
- tests/v1/simple_kv_offload
|
||||||
- tests/v1/worker
|
- tests/v1/worker
|
||||||
- tests/v1/kv_connector/unit
|
- tests/v1/kv_connector/unit
|
||||||
|
- tests/v1/ec_connector/unit
|
||||||
- tests/v1/metrics
|
- tests/v1/metrics
|
||||||
- tests/entrypoints/openai/correctness/test_lmeval.py
|
- tests/entrypoints/openai/correctness/test_lmeval.py
|
||||||
commands:
|
commands:
|
||||||
@@ -95,10 +102,17 @@ steps:
|
|||||||
- pytest -v -s v1/simple_kv_offload
|
- pytest -v -s v1/simple_kv_offload
|
||||||
- pytest -v -s v1/worker
|
- pytest -v -s v1/worker
|
||||||
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
- pytest -v -s -m 'not cpu_test' v1/kv_connector/unit
|
||||||
|
- pytest -v -s -m 'not cpu_test' v1/ec_connector/unit
|
||||||
- pytest -v -s -m 'not cpu_test' v1/metrics
|
- pytest -v -s -m 'not cpu_test' v1/metrics
|
||||||
# Integration test for streaming correctness (requires special branch).
|
# Integration test for streaming correctness (requires special branch).
|
||||||
- pip install -U git+https://github.com/robertgshaw2-redhat/lm-evaluation-harness.git@streaming-api
|
- pip install -U git+https://github.com/vllm-project/lm-evaluation-harness.git@streaming-api
|
||||||
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
- pytest -v -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 75
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: V1 Others (CPU)
|
- label: V1 Others (CPU)
|
||||||
key: v1-others-cpu
|
key: v1-others-cpu
|
||||||
@@ -160,7 +174,7 @@ steps:
|
|||||||
|
|
||||||
- label: Regression
|
- label: Regression
|
||||||
key: regression
|
key: regression
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 30
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
@@ -176,14 +190,14 @@ steps:
|
|||||||
- vllm/v1/
|
- vllm/v1/
|
||||||
- tests/test_regression
|
- tests/test_regression
|
||||||
commands:
|
commands:
|
||||||
- pip install modelscope
|
- pip install 'modelscope<1.38'
|
||||||
- pytest -v -s test_regression.py
|
- pytest -v -s test_regression.py
|
||||||
working_dir: "/vllm-workspace/tests" # optional
|
working_dir: "/vllm-workspace/tests" # optional
|
||||||
|
|
||||||
- label: Examples
|
- label: Examples
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: examples
|
key: examples
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 40
|
||||||
working_dir: "/vllm-workspace/examples"
|
working_dir: "/vllm-workspace/examples"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/entrypoints
|
- vllm/entrypoints
|
||||||
@@ -212,10 +226,20 @@ steps:
|
|||||||
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 2048
|
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 2048
|
||||||
# https://github.com/vllm-project/vllm/pull/26682 uses slightly more memory in PyTorch 2.9+ causing this test to OOM in 1xL4 GPU
|
# https://github.com/vllm-project/vllm/pull/26682 uses slightly more memory in PyTorch 2.9+ causing this test to OOM in 1xL4 GPU
|
||||||
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536
|
- python3 features/speculative_decoding/spec_decode_offline.py --test --method eagle3 --num_spec_tokens 3 --dataset-name hf --dataset-path philschmid/mt-bench --num-prompts 80 --temp 0 --top-p 1.0 --top-k -1 --tp 1 --enable-chunked-prefill --max-model-len 1536
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/entrypoints
|
||||||
|
- vllm/multimodal
|
||||||
|
- examples/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: Metrics, Tracing (2 GPUs)
|
- label: Metrics, Tracing (2 GPUs)
|
||||||
key: metrics-tracing-2-gpus
|
key: metrics-tracing-2-gpus
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 25
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
@@ -238,6 +262,12 @@ steps:
|
|||||||
'opentelemetry-exporter-otlp>=1.26.0' \
|
'opentelemetry-exporter-otlp>=1.26.0' \
|
||||||
'opentelemetry-semantic-conventions-ai>=0.4.1'"
|
'opentelemetry-semantic-conventions-ai>=0.4.1'"
|
||||||
- pytest -v -s v1/tracing
|
- pytest -v -s v1/tracing
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_2
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
optional: true
|
||||||
|
|
||||||
- label: Python-only Installation
|
- label: Python-only Installation
|
||||||
key: python-only-installation
|
key: python-only-installation
|
||||||
@@ -250,11 +280,21 @@ steps:
|
|||||||
- setup.py
|
- setup.py
|
||||||
commands:
|
commands:
|
||||||
- bash standalone_tests/python_only_compile.sh
|
- bash standalone_tests/python_only_compile.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 45
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- tests/standalone_tests/python_only_compile.sh
|
||||||
|
- setup.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: Async Engine, Inputs, Utils, Worker
|
- label: Async Engine, Inputs, Utils, Worker
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: async-engine-inputs-utils-worker
|
key: async-engine-inputs-utils-worker
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 25
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/assets/
|
- vllm/assets/
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
@@ -281,7 +321,7 @@ steps:
|
|||||||
key: async-engine-inputs-utils-worker-config-cpu
|
key: async-engine-inputs-utils-worker-config-cpu
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-cpu
|
- image-build-cpu
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 65
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/assets/
|
- vllm/assets/
|
||||||
- vllm/config/
|
- vllm/config/
|
||||||
@@ -313,6 +353,7 @@ steps:
|
|||||||
- tests/test_outputs.py
|
- tests/test_outputs.py
|
||||||
- tests/test_pooling_params.py
|
- tests/test_pooling_params.py
|
||||||
- tests/test_ray_env.py
|
- tests/test_ray_env.py
|
||||||
|
- tests/test_sampling_params.py
|
||||||
- tests/multimodal
|
- tests/multimodal
|
||||||
- tests/renderers
|
- tests/renderers
|
||||||
- tests/standalone_tests/lazy_imports.py
|
- tests/standalone_tests/lazy_imports.py
|
||||||
@@ -330,9 +371,10 @@ steps:
|
|||||||
- pytest -v -s test_outputs.py
|
- pytest -v -s test_outputs.py
|
||||||
- pytest -v -s test_pooling_params.py
|
- pytest -v -s test_pooling_params.py
|
||||||
- pytest -v -s test_ray_env.py
|
- pytest -v -s test_ray_env.py
|
||||||
|
- pytest -v -s test_sampling_params.py
|
||||||
- pytest -v -s -m 'cpu_test' multimodal
|
- pytest -v -s -m 'cpu_test' multimodal
|
||||||
- pytest -v -s renderers
|
- pytest -v -s renderers
|
||||||
- pytest -v -s reasoning --ignore=reasoning/test_seedoss_reasoning_parser.py --ignore=reasoning/test_glm4_moe_reasoning_parser.py
|
- pytest -v -s reasoning
|
||||||
- pytest -v -s tool_parsers
|
- pytest -v -s tool_parsers
|
||||||
- pytest -v -s tokenizers_
|
- pytest -v -s tokenizers_
|
||||||
- pytest -v -s parser
|
- pytest -v -s parser
|
||||||
@@ -341,7 +383,7 @@ steps:
|
|||||||
|
|
||||||
- label: Batch Invariance (A100)
|
- label: Batch Invariance (A100)
|
||||||
key: batch-invariance-a100
|
key: batch-invariance-a100
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
device: a100
|
device: a100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -355,7 +397,7 @@ steps:
|
|||||||
|
|
||||||
- label: Batch Invariance (H100)
|
- label: Batch Invariance (H100)
|
||||||
key: batch-invariance-h100
|
key: batch-invariance-h100
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
device: h100
|
device: h100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -371,7 +413,7 @@ steps:
|
|||||||
|
|
||||||
- label: Batch Invariance (B200)
|
- label: Batch Invariance (B200)
|
||||||
key: batch-invariance-b200
|
key: batch-invariance-b200
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 35
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/attention
|
- vllm/v1/attention
|
||||||
@@ -390,7 +432,7 @@ steps:
|
|||||||
- label: Acceptance Length Test (Large Models) # optional
|
- label: Acceptance Length Test (Large Models) # optional
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: acceptance-length-test-large-models
|
key: acceptance-length-test-large-models
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 20
|
||||||
gpu: h100
|
gpu: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_gpus: 1
|
num_gpus: 1
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Model Executor
|
- label: Model Executor
|
||||||
key: model-executor
|
key: model-executor
|
||||||
timeout_in_minutes: 35
|
timeout_in_minutes: 45
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/engine/arg_utils.py
|
- vllm/engine/arg_utils.py
|
||||||
- vllm/config/model.py
|
- vllm/config/model.py
|
||||||
@@ -23,3 +23,16 @@ steps:
|
|||||||
# calls that the signal method cannot interrupt.
|
# calls that the signal method cannot interrupt.
|
||||||
- pytest -v -s model_executor -m '(not slow_test)' --timeout=900 --timeout-method=thread
|
- pytest -v -s model_executor -m '(not slow_test)' --timeout=900 --timeout-method=thread
|
||||||
- pytest -v -s entrypoints/openai/completion/test_tensorizer_entrypoint.py --timeout=900 --timeout-method=thread
|
- pytest -v -s entrypoints/openai/completion/test_tensorizer_entrypoint.py --timeout=900 --timeout-method=thread
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/engine/arg_utils.py
|
||||||
|
- vllm/config/model.py
|
||||||
|
- vllm/model_executor
|
||||||
|
- tests/model_executor
|
||||||
|
- tests/entrypoints/openai/completion/test_tensorizer_entrypoint.py
|
||||||
|
- vllm/_aiter_ops.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ steps:
|
|||||||
- label: Model Runner V2 Core Tests
|
- label: Model Runner V2 Core Tests
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: model-runner-v2-core-tests
|
key: model-runner-v2-core-tests
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 35
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/worker/gpu/
|
- vllm/v1/worker/gpu/
|
||||||
- vllm/v1/worker/gpu_worker.py
|
- vllm/v1/worker/gpu_worker.py
|
||||||
@@ -18,9 +18,7 @@ steps:
|
|||||||
- set -x
|
- set -x
|
||||||
- export VLLM_USE_V2_MODEL_RUNNER=1
|
- export VLLM_USE_V2_MODEL_RUNNER=1
|
||||||
- pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics"
|
- pytest -v -s v1/engine/test_llm_engine.py -k "not test_engine_metrics"
|
||||||
# This requires eager until we sort out CG correctness issues.
|
- pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram"
|
||||||
# TODO: remove ENFORCE_EAGER here after https://github.com/vllm-project/vllm/pull/32936 is merged.
|
|
||||||
- ENFORCE_EAGER=1 pytest -v -s v1/e2e/general/test_async_scheduling.py -k "not ngram"
|
|
||||||
- pytest -v -s v1/e2e/general/test_context_length.py
|
- pytest -v -s v1/e2e/general/test_context_length.py
|
||||||
- pytest -v -s v1/e2e/general/test_min_tokens.py
|
- pytest -v -s v1/e2e/general/test_min_tokens.py
|
||||||
# Temporary hack filter to exclude ngram spec decoding based tests.
|
# Temporary hack filter to exclude ngram spec decoding based tests.
|
||||||
@@ -29,7 +27,7 @@ steps:
|
|||||||
- label: Model Runner V2 Examples
|
- label: Model Runner V2 Examples
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: model-runner-v2-examples
|
key: model-runner-v2-examples
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 35
|
||||||
working_dir: "/vllm-workspace/examples"
|
working_dir: "/vllm-workspace/examples"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/worker/gpu/
|
- vllm/v1/worker/gpu/
|
||||||
@@ -65,7 +63,7 @@ steps:
|
|||||||
|
|
||||||
- label: Model Runner V2 Distributed (2 GPUs)
|
- label: Model Runner V2 Distributed (2 GPUs)
|
||||||
key: model-runner-v2-distributed-2-gpus
|
key: model-runner-v2-distributed-2-gpus
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -86,7 +84,7 @@ steps:
|
|||||||
|
|
||||||
- label: Model Runner V2 Pipeline Parallelism (4 GPUs)
|
- label: Model Runner V2 Pipeline Parallelism (4 GPUs)
|
||||||
key: model-runner-v2-pipeline-parallelism-4-gpus
|
key: model-runner-v2-pipeline-parallelism-4-gpus
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 50
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -4,9 +4,8 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Basic Models Tests (Initialization)
|
- label: Basic Models Tests (Initialization)
|
||||||
key: basic-models-tests-initialization
|
key: basic-models-tests-initialization
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 25
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
torch_nightly: true
|
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/test_initialization.py
|
- tests/models/test_initialization.py
|
||||||
@@ -14,13 +13,11 @@ steps:
|
|||||||
commands:
|
commands:
|
||||||
# Run a subset of model initialization tests
|
# Run a subset of model initialization tests
|
||||||
- pytest -v -s models/test_initialization.py::test_can_initialize_small_subset
|
- pytest -v -s models/test_initialization.py::test_can_initialize_small_subset
|
||||||
mirror:
|
|
||||||
torch_nightly: {}
|
|
||||||
|
|
||||||
- label: Basic Models Tests (Extra Initialization) %N
|
- label: Basic Models Tests (Extra Initialization) %N
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: basic-models-tests-extra-initialization
|
key: basic-models-tests-extra-initialization
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 100
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/models/
|
- vllm/model_executor/models/
|
||||||
- tests/models/test_initialization.py
|
- tests/models/test_initialization.py
|
||||||
@@ -30,31 +27,35 @@ steps:
|
|||||||
# subset of supported models (the complement of the small subset in the above
|
# subset of supported models (the complement of the small subset in the above
|
||||||
# test.) Also run if model initialization test file is modified
|
# test.) Also run if model initialization test file is modified
|
||||||
- pytest -v -s models/test_initialization.py -k 'not test_can_initialize_small_subset' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
- pytest -v -s models/test_initialization.py -k 'not test_can_initialize_small_subset' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
||||||
parallelism: 2
|
parallelism: 4
|
||||||
mirror:
|
|
||||||
torch_nightly: {}
|
|
||||||
|
|
||||||
- label: Basic Models Tests (Other)
|
- label: Basic Models Tests (Other)
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: basic-models-tests-other
|
key: basic-models-tests-other
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 35
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/test_terratorch.py
|
- tests/models/test_terratorch.py
|
||||||
- tests/models/test_transformers.py
|
- tests/models/transformers/test_backend.py
|
||||||
- tests/models/test_registry.py
|
- tests/models/test_registry.py
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s models/test_terratorch.py models/test_transformers.py models/test_registry.py
|
- pytest -v -s models/test_terratorch.py models/transformers/test_backend.py models/test_registry.py
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: Basic Models Test (Other CPU) # 5min
|
- label: Basic Models Test (Other CPU) # 5min
|
||||||
key: basic-models-test-other-cpu
|
key: basic-models-test-other-cpu
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-cpu
|
- image-build-cpu
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 20
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/test_utils.py
|
- tests/models/test_utils.py
|
||||||
- tests/models/test_vision.py
|
- tests/models/test_vision.py
|
||||||
|
- tests/models/transformers/fusers/
|
||||||
device: cpu-small
|
device: cpu-small
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s models/test_utils.py models/test_vision.py
|
- pytest -v -s models/test_utils.py models/test_vision.py models/transformers/fusers/
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Distributed Model Tests (2 GPUs)
|
- label: Distributed Model Tests (2 GPUs)
|
||||||
key: distributed-model-tests-2-gpus
|
key: distributed-model-tests-2-gpus
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 60
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -17,7 +17,7 @@ steps:
|
|||||||
- TARGET_TEST_SUITE=L4 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
|
- TARGET_TEST_SUITE=L4 pytest basic_correctness/ -v -s -m 'distributed(num_gpus=2)'
|
||||||
- CUDA_VISIBLE_DEVICES=0,1 pytest -v -s model_executor/model_loader/test_sharded_state_loader.py -m '(not slow_test)'
|
- CUDA_VISIBLE_DEVICES=0,1 pytest -v -s model_executor/model_loader/test_sharded_state_loader.py -m '(not slow_test)'
|
||||||
# Avoid importing model tests that cause CUDA reinitialization error
|
# Avoid importing model tests that cause CUDA reinitialization error
|
||||||
- pytest models/test_transformers.py -v -s -m 'distributed(num_gpus=2)'
|
- pytest models/transformers/test_backend.py -v -s -m 'distributed(num_gpus=2)'
|
||||||
- pytest models/language -v -s -m 'distributed(num_gpus=2)'
|
- pytest models/language -v -s -m 'distributed(num_gpus=2)'
|
||||||
- pytest models/multimodal/generation/test_phi4siglip.py -v -s -m 'distributed(num_gpus=2)'
|
- pytest models/multimodal/generation/test_phi4siglip.py -v -s -m 'distributed(num_gpus=2)'
|
||||||
- pytest models/multimodal -v -s -m 'distributed(num_gpus=2)' --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_phi4siglip.py
|
- pytest models/multimodal -v -s -m 'distributed(num_gpus=2)' --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_phi4siglip.py
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Language Models Tests (Standard)
|
- label: Language Models Tests (Standard)
|
||||||
key: language-models-tests-standard
|
key: language-models-tests-standard
|
||||||
timeout_in_minutes: 25
|
timeout_in_minutes: 30
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -14,11 +14,14 @@ steps:
|
|||||||
- pip freeze | grep -E 'torch'
|
- pip freeze | grep -E 'torch'
|
||||||
- pytest -v -s models/language -m 'core_model and (not slow_test)'
|
- pytest -v -s models/language -m 'core_model and (not slow_test)'
|
||||||
mirror:
|
mirror:
|
||||||
torch_nightly: {}
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: Language Models Tests (Extra Standard) %N
|
- label: Language Models Tests (Extra Standard) %N
|
||||||
key: language-models-tests-extra-standard
|
key: language-models-tests-extra-standard
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/models/
|
- vllm/model_executor/models/
|
||||||
- tests/models/language/pooling/test_embedding.py
|
- tests/models/language/pooling/test_embedding.py
|
||||||
@@ -31,11 +34,25 @@ steps:
|
|||||||
- pytest -v -s models/language -m 'core_model and slow_test' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
- pytest -v -s models/language -m 'core_model and slow_test' --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
torch_nightly: {}
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/model_executor/models/
|
||||||
|
- vllm/model_executor/model_loader/
|
||||||
|
- vllm/model_executor/layers/
|
||||||
|
- vllm/v1/attention/backends/
|
||||||
|
- vllm/v1/attention/selector.py
|
||||||
|
- tests/models/language/pooling/test_embedding.py
|
||||||
|
- tests/models/language/generation/test_common.py
|
||||||
|
- tests/models/language/pooling/test_classification.py
|
||||||
|
- vllm/_aiter_ops.py
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: Language Models Tests (Hybrid) %N
|
- label: Language Models Tests (Hybrid) %N
|
||||||
key: language-models-tests-hybrid
|
key: language-models-tests-hybrid
|
||||||
timeout_in_minutes: 75
|
timeout_in_minutes: 65
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/language/generation
|
- tests/models/language/generation
|
||||||
@@ -48,10 +65,9 @@ steps:
|
|||||||
- pytest -v -s models/language/generation -m hybrid_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
- pytest -v -s models/language/generation -m hybrid_model --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
||||||
parallelism: 2
|
parallelism: 2
|
||||||
mirror:
|
mirror:
|
||||||
torch_nightly: {}
|
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 70
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
commands:
|
commands:
|
||||||
@@ -62,7 +78,7 @@ steps:
|
|||||||
- label: Language Models Test (Extended Generation) # 80min
|
- label: Language Models Test (Extended Generation) # 80min
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: language-models-test-extended-generation
|
key: language-models-test-extended-generation
|
||||||
timeout_in_minutes: 110
|
timeout_in_minutes: 65
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -76,7 +92,7 @@ steps:
|
|||||||
|
|
||||||
- label: Language Models Test (PPL)
|
- label: Language Models Test (PPL)
|
||||||
key: language-models-test-ppl
|
key: language-models-test-ppl
|
||||||
timeout_in_minutes: 110
|
timeout_in_minutes: 30
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -88,7 +104,7 @@ steps:
|
|||||||
- label: Language Models Test (Extended Pooling) # 36min
|
- label: Language Models Test (Extended Pooling) # 36min
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: language-models-test-extended-pooling
|
key: language-models-test-extended-pooling
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 70
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
@@ -104,7 +120,7 @@ steps:
|
|||||||
|
|
||||||
- label: Language Models Test (MTEB)
|
- label: Language Models Test (MTEB)
|
||||||
key: language-models-test-mteb
|
key: language-models-test-mteb
|
||||||
timeout_in_minutes: 110
|
timeout_in_minutes: 45
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -10,7 +10,6 @@ steps:
|
|||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen2"
|
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen2"
|
||||||
- pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model
|
- pytest -v -s models/multimodal/generation/test_ultravox.py -m core_model
|
||||||
mirror:
|
mirror:
|
||||||
@@ -21,16 +20,15 @@ steps:
|
|||||||
|
|
||||||
- label: "Multi-Modal Models (Standard) 2: qwen3 + gemma"
|
- label: "Multi-Modal Models (Standard) 2: qwen3 + gemma"
|
||||||
key: multi-modal-models-standard-2-qwen3-gemma
|
key: multi-modal-models-standard-2-qwen3-gemma
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 50
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen3 or gemma"
|
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "qwen3 or gemma"
|
||||||
|
- pytest -v -s models/multimodal/generation/test_mm_prefix_lm.py -m core_model
|
||||||
- pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model
|
- pytest -v -s models/multimodal/generation/test_qwen2_5_vl.py -m core_model
|
||||||
- pytest -v -s models/multimodal/generation/test_vit_cudagraph.py -m core_model
|
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
@@ -40,12 +38,11 @@ steps:
|
|||||||
- label: "Multi-Modal Models (Standard) 3: llava + qwen2_vl"
|
- label: "Multi-Modal Models (Standard) 3: llava + qwen2_vl"
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: multi-modal-models-standard-3-llava-qwen2-vl
|
key: multi-modal-models-standard-3-llava-qwen2-vl
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "not qwen2 and not qwen3 and not gemma"
|
- pytest -v -s models/multimodal/generation/test_common.py -m core_model -k "not qwen2 and not qwen3 and not gemma"
|
||||||
- pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model
|
- pytest -v -s models/multimodal/generation/test_qwen2_vl.py -m core_model
|
||||||
mirror:
|
mirror:
|
||||||
@@ -57,46 +54,50 @@ steps:
|
|||||||
- label: "Multi-Modal Models (Standard) 4: other + whisper"
|
- label: "Multi-Modal Models (Standard) 4: other + whisper"
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: multi-modal-models-standard-4-other-whisper
|
key: multi-modal-models-standard-4-other-whisper
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 50
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
- pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_mm_prefix_lm.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/generation/test_vit_cudagraph.py --ignore models/multimodal/processing
|
||||||
- pytest -v -s models/multimodal -m core_model --ignore models/multimodal/generation/test_common.py --ignore models/multimodal/generation/test_ultravox.py --ignore models/multimodal/generation/test_qwen2_5_vl.py --ignore models/multimodal/generation/test_qwen2_vl.py --ignore models/multimodal/generation/test_whisper.py --ignore models/multimodal/generation/test_memory_leak.py --ignore models/multimodal/processing
|
- pytest -v -s models/multimodal/generation/test_vit_cudagraph.py -m core_model
|
||||||
- pytest models/multimodal/generation/test_memory_leak.py -m core_model
|
- pytest models/multimodal/generation/test_memory_leak.py -m core_model
|
||||||
- cd .. && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s tests/models/multimodal/generation/test_whisper.py -m core_model # Otherwise, mp_method="spawn" doesn't work
|
- cd .. && VLLM_WORKER_MULTIPROC_METHOD=spawn pytest -v -s tests/models/multimodal/generation/test_whisper.py -m core_model # Otherwise, mp_method="spawn" doesn't work
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: Multi-Modal Processor (CPU)
|
- label: Multi-Modal Processor (CPU) %N
|
||||||
key: multi-modal-processor-cpu
|
key: multi-modal-processor-cpu
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-cpu
|
- image-build-cpu
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 125
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
- tests/models/registry.py
|
- tests/models/registry.py
|
||||||
device: cpu-medium
|
device: cpu-medium
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
- pytest -v -s models/multimodal/processing --ignore models/multimodal/processing/test_tensor_schema.py --num-shards=$$BUILDKITE_PARALLEL_JOB_COUNT --shard-id=$$BUILDKITE_PARALLEL_JOB
|
||||||
- pytest -v -s models/multimodal/processing --ignore models/multimodal/processing/test_tensor_schema.py
|
parallelism: 4
|
||||||
|
|
||||||
- label: Multi-Modal Processor # 44min
|
- label: Multi-Modal Processor # 44min
|
||||||
key: multi-modal-processor
|
key: multi-modal-processor
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 65
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal
|
- tests/models/multimodal
|
||||||
- tests/models/registry.py
|
- tests/models/registry.py
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/processing/test_tensor_schema.py
|
- pytest -v -s models/multimodal/processing/test_tensor_schema.py
|
||||||
|
|
||||||
- label: Multi-Modal Accuracy Eval (Small Models) # 50min
|
- label: Multi-Modal Accuracy Eval (Small Models) # 50min
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: multi-modal-accuracy-eval-small-models
|
key: multi-modal-accuracy-eval-small-models
|
||||||
timeout_in_minutes: 70
|
timeout_in_minutes: 30
|
||||||
working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
|
working_dir: "/vllm-workspace/.buildkite/lm-eval-harness"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/multimodal/
|
- vllm/multimodal/
|
||||||
@@ -104,6 +105,17 @@ steps:
|
|||||||
- vllm/v1/core/
|
- vllm/v1/core/
|
||||||
commands:
|
commands:
|
||||||
- pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-mm-small.txt --tp-size=1
|
- pytest -s -v test_lm_eval_correctness.py --config-list-file=configs/models-mm-small.txt --tp-size=1
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/multimodal/
|
||||||
|
- vllm/inputs/
|
||||||
|
- vllm/v1/core/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
- vllm/model_executor/model_loader/
|
||||||
|
|
||||||
- label: Multi-Modal Models (Extended Generation 1)
|
- label: Multi-Modal Models (Extended Generation 1)
|
||||||
key: multi-modal-models-extended-generation-1
|
key: multi-modal-models-extended-generation-1
|
||||||
@@ -113,7 +125,6 @@ steps:
|
|||||||
- tests/models/multimodal/generation
|
- tests/models/multimodal/generation
|
||||||
- tests/models/multimodal/test_mapping.py
|
- tests/models/multimodal/test_mapping.py
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/generation -m 'not core_model' --ignore models/multimodal/generation/test_common.py
|
- pytest -v -s models/multimodal/generation -m 'not core_model' --ignore models/multimodal/generation/test_common.py
|
||||||
- pytest -v -s models/multimodal/test_mapping.py
|
- pytest -v -s models/multimodal/test_mapping.py
|
||||||
mirror:
|
mirror:
|
||||||
@@ -130,7 +141,6 @@ steps:
|
|||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal/generation
|
- tests/models/multimodal/generation
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=0) and not core_model'
|
- pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=0) and not core_model'
|
||||||
|
|
||||||
- label: Multi-Modal Models (Extended Generation 3)
|
- label: Multi-Modal Models (Extended Generation 3)
|
||||||
@@ -141,7 +151,6 @@ steps:
|
|||||||
- vllm/
|
- vllm/
|
||||||
- tests/models/multimodal/generation
|
- tests/models/multimodal/generation
|
||||||
commands:
|
commands:
|
||||||
- pip install git+https://github.com/TIGER-AI-Lab/Mantis.git
|
|
||||||
- pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=1) and not core_model'
|
- pytest -v -s models/multimodal/generation/test_common.py -m 'split(group=1) and not core_model'
|
||||||
|
|
||||||
- label: Multi-Modal Models (Extended Pooling)
|
- label: Multi-Modal Models (Extended Pooling)
|
||||||
@@ -156,7 +165,7 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 75
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Plugin Tests (2 GPUs)
|
- label: Plugin Tests (2 GPUs)
|
||||||
key: plugin-tests-2-gpus
|
key: plugin-tests-2-gpus
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 35
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -27,12 +27,21 @@ steps:
|
|||||||
- pip install -e ./plugins/bge_m3_sparse_plugin
|
- pip install -e ./plugins/bge_m3_sparse_plugin
|
||||||
- pytest -v -s plugins_tests/test_bge_m3_sparse_io_processor_plugins.py
|
- pytest -v -s plugins_tests/test_bge_m3_sparse_io_processor_plugins.py
|
||||||
- pip uninstall bge_m3_sparse_plugin -y
|
- pip uninstall bge_m3_sparse_plugin -y
|
||||||
|
# test colbert_query io_processor plugin
|
||||||
|
- pip install -e ./plugins/colbert_query_plugin
|
||||||
|
- pytest -v -s plugins_tests/test_colbert_query_io_processor_plugins.py
|
||||||
|
- pip uninstall colbert_query_plugin -y
|
||||||
# end io_processor plugins test
|
# end io_processor plugins test
|
||||||
# begin stat_logger plugins test
|
# begin stat_logger plugins test
|
||||||
- pip install -e ./plugins/vllm_add_dummy_stat_logger
|
- pip install -e ./plugins/vllm_add_dummy_stat_logger
|
||||||
- pytest -v -s plugins_tests/test_stats_logger_plugins.py
|
- pytest -v -s plugins_tests/test_stats_logger_plugins.py
|
||||||
- pip uninstall dummy_stat_logger -y
|
- pip uninstall dummy_stat_logger -y
|
||||||
# end stat_logger plugins test
|
# end stat_logger plugins test
|
||||||
|
# begin endpoint plugins test
|
||||||
|
- pip install -e ./plugins/vllm_add_dummy_endpoint_plugin
|
||||||
|
- pytest -v -s plugins_tests/test_endpoint_plugins.py
|
||||||
|
- pip uninstall vllm_add_dummy_endpoint_plugin -y
|
||||||
|
# end endpoint plugins test
|
||||||
# other tests continue here:
|
# other tests continue here:
|
||||||
- pytest -v -s plugins_tests/test_scheduler_plugins.py
|
- pytest -v -s plugins_tests/test_scheduler_plugins.py
|
||||||
- pip install -e ./plugins/vllm_add_dummy_model
|
- pip install -e ./plugins/vllm_add_dummy_model
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ steps:
|
|||||||
- label: PyTorch Compilation Unit Tests
|
- label: PyTorch Compilation Unit Tests
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: pytorch-compilation-unit-tests
|
key: pytorch-compilation-unit-tests
|
||||||
timeout_in_minutes: 10
|
timeout_in_minutes: 90
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/__init__.py
|
- vllm/__init__.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
@@ -78,7 +78,7 @@ steps:
|
|||||||
|
|
||||||
- label: PyTorch Compilation Passes Unit Tests
|
- label: PyTorch Compilation Passes Unit Tests
|
||||||
key: pytorch-compilation-passes-unit-tests
|
key: pytorch-compilation-passes-unit-tests
|
||||||
timeout_in_minutes: 20
|
timeout_in_minutes: 45
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/__init__.py
|
- vllm/__init__.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
@@ -107,10 +107,16 @@ steps:
|
|||||||
- tests/compile/passes
|
- tests/compile/passes
|
||||||
commands:
|
commands:
|
||||||
- pytest -s -v compile/passes --ignore compile/passes/distributed
|
- pytest -s -v compile/passes --ignore compile/passes/distributed
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
timeout_in_minutes: 65
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
|
||||||
- label: PyTorch Fullgraph Smoke Test
|
- label: PyTorch Fullgraph Smoke Test
|
||||||
key: pytorch-fullgraph-smoke-test
|
key: pytorch-fullgraph-smoke-test
|
||||||
timeout_in_minutes: 35
|
timeout_in_minutes: 60
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/__init__.py
|
- vllm/__init__.py
|
||||||
- vllm/_aiter_ops.py
|
- vllm/_aiter_ops.py
|
||||||
@@ -146,7 +152,7 @@ steps:
|
|||||||
|
|
||||||
- label: PyTorch Fullgraph
|
- label: PyTorch Fullgraph
|
||||||
key: pytorch-fullgraph
|
key: pytorch-fullgraph
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/__init__.py
|
- vllm/__init__.py
|
||||||
@@ -189,3 +195,11 @@ steps:
|
|||||||
- requirements/test/nightly-torch.txt
|
- requirements/test/nightly-torch.txt
|
||||||
commands:
|
commands:
|
||||||
- bash standalone_tests/pytorch_nightly_dependency.sh
|
- bash standalone_tests/pytorch_nightly_dependency.sh
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_1
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- requirements/test/nightly-torch.txt
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Quantization
|
- label: Quantization
|
||||||
key: quantization
|
key: quantization
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 60
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- csrc/
|
- csrc/
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
@@ -23,7 +23,7 @@ steps:
|
|||||||
|
|
||||||
- label: Quantized Fusions
|
- label: Quantized Fusions
|
||||||
key: quantized-fusions
|
key: quantized-fusions
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- tests/fusion
|
- tests/fusion
|
||||||
- vllm/model_executor/layers/fusion
|
- vllm/model_executor/layers/fusion
|
||||||
@@ -35,7 +35,7 @@ steps:
|
|||||||
|
|
||||||
- label: Quantized MoE Test (B200)
|
- label: Quantized MoE Test (B200)
|
||||||
key: quantized-moe-test-b200
|
key: quantized-moe-test-b200
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 120
|
||||||
working_dir: "/vllm-workspace/"
|
working_dir: "/vllm-workspace/"
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -53,7 +53,7 @@ steps:
|
|||||||
|
|
||||||
- label: Quantized Models Test
|
- label: Quantized Models Test
|
||||||
key: quantized-models-test
|
key: quantized-models-test
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 50
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/layers/quantization
|
- vllm/model_executor/layers/quantization
|
||||||
- tests/models/quantization
|
- tests/models/quantization
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ depends_on:
|
|||||||
- image-build
|
- image-build
|
||||||
steps:
|
steps:
|
||||||
- label: Rust Frontend OpenAI Coverage
|
- label: Rust Frontend OpenAI Coverage
|
||||||
timeout_in_minutes: 90
|
timeout_in_minutes: 30
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -15,28 +15,30 @@ steps:
|
|||||||
- tests/utils.py
|
- tests/utils.py
|
||||||
- tests/benchmarks/test_serve_cli.py
|
- tests/benchmarks/test_serve_cli.py
|
||||||
- tests/entrypoints/openai/chat_completion/test_chat_completion.py
|
- tests/entrypoints/openai/chat_completion/test_chat_completion.py
|
||||||
# - tests/entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py
|
- tests/entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py
|
||||||
|
|
||||||
# - tests/entrypoints/openai/completion/test_prompt_validation.py
|
# - tests/entrypoints/openai/completion/test_prompt_validation.py
|
||||||
- tests/entrypoints/openai/completion/test_shutdown.py
|
- tests/entrypoints/openai/completion/test_shutdown.py
|
||||||
# - tests/entrypoints/openai/test_return_token_ids.py
|
- tests/entrypoints/openai/test_return_token_ids.py
|
||||||
# - tests/entrypoints/openai/test_uds.py
|
- tests/entrypoints/openai/test_uds.py
|
||||||
- tests/v1/sample/test_logprobs_e2e.py
|
- tests/v1/sample/test_logprobs_e2e.py
|
||||||
commands:
|
commands:
|
||||||
- export VLLM_USE_RUST_FRONTEND=1
|
- export VLLM_USE_RUST_FRONTEND=1
|
||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
- pytest -v -s benchmarks/test_serve_cli.py -k "not insecure and not (test_bench_serve and not test_bench_serve_chat)"
|
- pytest -v -s benchmarks/test_serve_cli.py -k "not insecure and not (test_bench_serve and not test_bench_serve_chat)"
|
||||||
- pytest -v -s entrypoints/openai/chat_completion/test_chat_completion.py
|
- pytest -v -s entrypoints/openai/chat_completion/test_chat_completion.py -k "not test_invalid_json_schema and not test_invalid_regex"
|
||||||
# - pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not invalid"
|
- pytest -v -s entrypoints/openai/chat_completion/test_chat_logit_bias_validation.py -k "not multiple"
|
||||||
|
|
||||||
# - pytest -v -s entrypoints/openai/completion/test_prompt_validation.py -k "not prompt_embeds"
|
# - pytest -v -s entrypoints/openai/completion/test_prompt_validation.py -k "not prompt_embeds"
|
||||||
- pytest -v -s entrypoints/openai/completion/test_shutdown.py -k "not engine_failure and not test_abort_timeout_exits_quickly"
|
- pytest -v -s entrypoints/openai/completion/test_shutdown.py -k "not engine_failure and not test_abort_timeout_exits_quickly"
|
||||||
# - pytest -v -s entrypoints/openai/test_return_token_ids.py
|
# test_comparison streams differently: Rust emits a separate first (prompt_token_ids) chunk and
|
||||||
# - pytest -v -s entrypoints/openai/test_uds.py
|
# finish chunk without logprobs, while the test reads `logprobs.tokens` on every chunk.
|
||||||
|
- pytest -v -s entrypoints/openai/test_return_token_ids.py -k "not test_comparison"
|
||||||
|
- pytest -v -s entrypoints/openai/test_uds.py
|
||||||
- pytest -v -s v1/sample/test_logprobs_e2e.py -k "test_prompt_logprobs_e2e_server"
|
- pytest -v -s v1/sample/test_logprobs_e2e.py -k "test_prompt_logprobs_e2e_server"
|
||||||
|
|
||||||
- label: Rust Frontend Serve/Admin Coverage
|
- label: Rust Frontend Serve/Admin Coverage
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 25
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -45,22 +47,27 @@ steps:
|
|||||||
- vllm/entrypoints/serve/
|
- vllm/entrypoints/serve/
|
||||||
- vllm/v1/engine/
|
- vllm/v1/engine/
|
||||||
- tests/utils.py
|
- tests/utils.py
|
||||||
# - tests/entrypoints/serve/dev/rpc/test_collective_rpc.py
|
- tests/entrypoints/serve/dev/rpc/test_collective_rpc.py
|
||||||
- tests/entrypoints/serve/disagg/test_serving_tokens.py
|
- tests/entrypoints/scale_out/token_in_token_out/test_serving_tokens.py
|
||||||
- tests/entrypoints/serve/instrumentator/test_basic.py
|
- tests/entrypoints/serve/instrumentator/test_basic.py
|
||||||
- tests/entrypoints/serve/instrumentator/test_metrics.py
|
- tests/entrypoints/serve/instrumentator/test_metrics.py
|
||||||
# - tests/entrypoints/serve/dev/test_sleep.py
|
# - tests/entrypoints/serve/dev/test_sleep.py
|
||||||
|
- tests/entrypoints/serve/tokenize/test_tokenization.py
|
||||||
commands:
|
commands:
|
||||||
- export VLLM_USE_RUST_FRONTEND=1
|
- export VLLM_USE_RUST_FRONTEND=1
|
||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
# - pytest -v -s entrypoints/serve/dev/rpc/test_collective_rpc.py
|
- PYTHONPATH=/vllm-workspace pytest -v -s entrypoints/serve/dev/rpc/test_collective_rpc.py
|
||||||
|
# server_load can be flaky under the Rust frontend; keep it excluded for now.
|
||||||
- pytest -v -s entrypoints/serve/instrumentator/test_basic.py -k "not show_version and not server_load"
|
- pytest -v -s entrypoints/serve/instrumentator/test_basic.py -k "not show_version and not server_load"
|
||||||
- pytest -v -s entrypoints/serve/disagg/test_serving_tokens.py -k "not stream and not lora and not test_generate_logprobs and not stop_string_workflow"
|
# test_generate_logprobs expects Python-style top_logprobs truncation (dedup sampled + cap at max(k, 1)).
|
||||||
|
- pytest -v -s entrypoints/scale_out/token_in_token_out/test_serving_tokens.py -k "not stream and not lora and not test_generate_logprobs and not stop_string_workflow"
|
||||||
- pytest -v -s entrypoints/serve/instrumentator/test_metrics.py -k "text and not show and not run_batch and not test_metrics_counts and not test_metrics_exist"
|
- pytest -v -s entrypoints/serve/instrumentator/test_metrics.py -k "text and not show and not run_batch and not test_metrics_counts and not test_metrics_exist"
|
||||||
# - pytest -v -s entrypoints/serve/dev/test_sleep.py
|
# - pytest -v -s entrypoints/serve/dev/test_sleep.py
|
||||||
|
# /tokenizer_info is not implemented in the Rust frontend (the CLI flag is accepted as a no-op).
|
||||||
|
- pytest -v -s entrypoints/serve/tokenize/test_tokenization.py -k "not tokenizer_info"
|
||||||
|
|
||||||
- label: Rust Frontend Core Correctness
|
- label: Rust Frontend Core Correctness
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -74,7 +81,7 @@ steps:
|
|||||||
- pytest -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
- pytest -s entrypoints/openai/correctness/test_lmeval.py::test_lm_eval_accuracy_v1_engine
|
||||||
|
|
||||||
- label: Rust Frontend Tool Use
|
- label: Rust Frontend Tool Use
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 25
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- rust/
|
- rust/
|
||||||
@@ -88,7 +95,7 @@ steps:
|
|||||||
- pytest -v -s tool_use --ignore=tool_use/mistral --models llama3.2 -k "not test_response_format_with_tool_choice_required and not test_parallel_tool_calls_false and not test_tool_call_and_choice"
|
- pytest -v -s tool_use --ignore=tool_use/mistral --models llama3.2 -k "not test_response_format_with_tool_choice_required and not test_parallel_tool_calls_false and not test_tool_call_and_choice"
|
||||||
|
|
||||||
- label: Rust Frontend Distributed
|
- label: Rust Frontend Distributed
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 25
|
||||||
num_devices: 4
|
num_devices: 4
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -99,9 +106,13 @@ steps:
|
|||||||
- vllm/v1/engine/
|
- vllm/v1/engine/
|
||||||
- vllm/v1/worker/
|
- vllm/v1/worker/
|
||||||
- tests/utils.py
|
- tests/utils.py
|
||||||
|
- tests/v1/distributed/test_external_lb_dp.py
|
||||||
|
- tests/v1/distributed/test_hybrid_lb_dp.py
|
||||||
- tests/v1/distributed/test_internal_lb_dp.py
|
- tests/v1/distributed/test_internal_lb_dp.py
|
||||||
commands:
|
commands:
|
||||||
- export VLLM_USE_RUST_FRONTEND=1
|
- export VLLM_USE_RUST_FRONTEND=1
|
||||||
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
- export VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||||
- export NCCL_CUMEM_HOST_ENABLE=0
|
- export NCCL_CUMEM_HOST_ENABLE=0
|
||||||
- TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_internal_lb_dp.py -k "not 4 and not server_info"
|
- TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_internal_lb_dp.py -k "not 4 and not server_info"
|
||||||
|
- TP_SIZE=1 DP_SIZE=2 pytest -v -s v1/distributed/test_external_lb_dp.py -k "not 4 and not server_info"
|
||||||
|
- TP_SIZE=1 DP_SIZE=4 pytest -v -s v1/distributed/test_hybrid_lb_dp.py -k "not 4 and not server_info"
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ steps:
|
|||||||
- label: Rust Frontend Cargo Style + Clippy
|
- label: Rust Frontend Cargo Style + Clippy
|
||||||
key: rust-frontend-cargo-style-clippy
|
key: rust-frontend-cargo-style-clippy
|
||||||
depends_on: []
|
depends_on: []
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: cpu-medium
|
device: cpu-medium
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -18,7 +18,7 @@ steps:
|
|||||||
- label: Rust Frontend Cargo Tests
|
- label: Rust Frontend Cargo Tests
|
||||||
key: rust-frontend-cargo-tests
|
key: rust-frontend-cargo-tests
|
||||||
depends_on: []
|
depends_on: []
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: cpu-medium
|
device: cpu-medium
|
||||||
no_plugin: true
|
no_plugin: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ steps:
|
|||||||
- label: Samplers Test
|
- label: Samplers Test
|
||||||
device: h200_35gb
|
device: h200_35gb
|
||||||
key: samplers-test
|
key: samplers-test
|
||||||
timeout_in_minutes: 75
|
timeout_in_minutes: 40
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/model_executor/layers
|
- vllm/model_executor/layers
|
||||||
- vllm/sampling_metadata.py
|
- vllm/sampling_metadata.py
|
||||||
@@ -19,7 +19,7 @@ steps:
|
|||||||
- VLLM_USE_FLASHINFER_SAMPLER=1 pytest -v -s samplers
|
- VLLM_USE_FLASHINFER_SAMPLER=1 pytest -v -s samplers
|
||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi250_1
|
device: mi325_1
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
commands:
|
commands:
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Spec Decode Eagle
|
- label: Spec Decode Eagle
|
||||||
key: spec-decode-eagle
|
key: spec-decode-eagle
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 25
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/spec_decode/
|
- vllm/v1/spec_decode/
|
||||||
@@ -12,10 +12,24 @@ steps:
|
|||||||
- tests/v1/e2e/spec_decode/
|
- tests/v1/e2e/spec_decode/
|
||||||
commands:
|
commands:
|
||||||
- pytest -v -s v1/e2e/spec_decode -k "eagle_correctness"
|
- pytest -v -s v1/e2e/spec_decode -k "eagle_correctness"
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi325_1
|
||||||
|
timeout_in_minutes: 60
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
source_file_dependencies:
|
||||||
|
- vllm/v1/spec_decode/
|
||||||
|
- vllm/v1/worker/gpu/spec_decode/
|
||||||
|
- vllm/model_executor/model_loader/
|
||||||
|
- vllm/v1/sample/
|
||||||
|
- vllm/model_executor/layers/
|
||||||
|
- tests/v1/e2e/spec_decode/
|
||||||
|
- vllm/platforms/rocm.py
|
||||||
|
|
||||||
- label: Spec Decode Eagle Nightly B200
|
- label: Spec Decode Eagle Nightly B200
|
||||||
key: spec-decode-eagle-nightly-b200
|
key: spec-decode-eagle-nightly-b200
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 25
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -27,7 +41,7 @@ steps:
|
|||||||
|
|
||||||
- label: Spec Decode Speculators + MTP
|
- label: Spec Decode Speculators + MTP
|
||||||
key: spec-decode-speculators-mtp
|
key: spec-decode-speculators-mtp
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/spec_decode/
|
- vllm/v1/spec_decode/
|
||||||
@@ -68,7 +82,7 @@ steps:
|
|||||||
|
|
||||||
- label: Spec Decode Ngram + Suffix
|
- label: Spec Decode Ngram + Suffix
|
||||||
key: spec-decode-ngram-suffix
|
key: spec-decode-ngram-suffix
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/spec_decode/
|
- vllm/v1/spec_decode/
|
||||||
@@ -79,7 +93,9 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 65
|
timeout_in_minutes: 55
|
||||||
|
# TODO(akaratza): Test after Torch >= 2.12 bump
|
||||||
|
soft_fail: true
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -93,7 +109,7 @@ steps:
|
|||||||
|
|
||||||
- label: Spec Decode Draft Model
|
- label: Spec Decode Draft Model
|
||||||
key: spec-decode-draft-model
|
key: spec-decode-draft-model
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 45
|
||||||
device: h200_18gb
|
device: h200_18gb
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
- vllm/v1/spec_decode/
|
- vllm/v1/spec_decode/
|
||||||
@@ -104,7 +120,7 @@ steps:
|
|||||||
mirror:
|
mirror:
|
||||||
amd:
|
amd:
|
||||||
device: mi325_1
|
device: mi325_1
|
||||||
timeout_in_minutes: 50
|
timeout_in_minutes: 55
|
||||||
depends_on:
|
depends_on:
|
||||||
- image-build-amd
|
- image-build-amd
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -118,7 +134,7 @@ steps:
|
|||||||
|
|
||||||
- label: Spec Decode Draft Model Nightly B200
|
- label: Spec Decode Draft Model Nightly B200
|
||||||
key: spec-decode-draft-model-nightly-b200
|
key: spec-decode-draft-model-nightly-b200
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 40
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
@@ -130,7 +146,7 @@ steps:
|
|||||||
|
|
||||||
- label: Speculators Correctness
|
- label: Speculators Correctness
|
||||||
key: speculators-correctness
|
key: speculators-correctness
|
||||||
timeout_in_minutes: 60
|
timeout_in_minutes: 30
|
||||||
device: h100
|
device: h100
|
||||||
optional: true
|
optional: true
|
||||||
num_devices: 1
|
num_devices: 1
|
||||||
@@ -143,7 +159,7 @@ steps:
|
|||||||
- pytest -v -s v1/spec_decode/test_speculators_correctness.py -m slow_test
|
- pytest -v -s v1/spec_decode/test_speculators_correctness.py -m slow_test
|
||||||
|
|
||||||
- label: Spec Decode MTP hybrid (B200)
|
- label: Spec Decode MTP hybrid (B200)
|
||||||
timeout_in_minutes: 30
|
timeout_in_minutes: 20
|
||||||
device: b200-k8s
|
device: b200-k8s
|
||||||
optional: true
|
optional: true
|
||||||
source_file_dependencies:
|
source_file_dependencies:
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ depends_on:
|
|||||||
steps:
|
steps:
|
||||||
- label: Weight Loading Multiple GPU # 33min
|
- label: Weight Loading Multiple GPU # 33min
|
||||||
key: weight-loading-multiple-gpu
|
key: weight-loading-multiple-gpu
|
||||||
timeout_in_minutes: 45
|
timeout_in_minutes: 50
|
||||||
working_dir: "/vllm-workspace/tests"
|
working_dir: "/vllm-workspace/tests"
|
||||||
num_devices: 2
|
num_devices: 2
|
||||||
optional: true
|
optional: true
|
||||||
@@ -13,6 +13,13 @@ steps:
|
|||||||
- tests/weight_loading
|
- tests/weight_loading
|
||||||
commands:
|
commands:
|
||||||
- bash weight_loading/run_model_weight_loading_test.sh -c weight_loading/models.txt
|
- bash weight_loading/run_model_weight_loading_test.sh -c weight_loading/models.txt
|
||||||
|
mirror:
|
||||||
|
amd:
|
||||||
|
device: mi300_2
|
||||||
|
depends_on:
|
||||||
|
- image-build-amd
|
||||||
|
commands:
|
||||||
|
- bash weight_loading/run_model_weight_loading_test.sh -c weight_loading/models-amd.txt
|
||||||
|
|
||||||
# - label: Weight Loading Multiple GPU - Large Models # optional
|
# - label: Weight Loading Multiple GPU - Large Models # optional
|
||||||
# working_dir: "/vllm-workspace/tests"
|
# working_dir: "/vllm-workspace/tests"
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
---
|
||||||
|
name: ci-fails-buildkite
|
||||||
|
description: Fetch and diagnose vLLM Buildkite CI failure logs. Use when investigating failing CI jobs on a PR or build, when the user pastes a buildkite.com URL, or asks to fetch/diagnose CI logs.
|
||||||
|
---
|
||||||
|
|
||||||
|
# Diagnosing vLLM Buildkite CI Failures
|
||||||
|
|
||||||
|
Buildkite logs are public; no login needed.
|
||||||
|
|
||||||
|
`.buildkite/scripts/ci-fetch-log.sh` saves each log as `ci-<build>-<job-name>.log`, stripped of timestamps and ANSI codes. Existing files are kept; set `CI_FETCH_LOG_FORCE=1` to refetch.
|
||||||
|
|
||||||
|
## Fetching logs
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# All failed jobs in a PR's latest build (current branch's PR if omitted):
|
||||||
|
.buildkite/scripts/ci-fetch-log.sh --pr <PR>
|
||||||
|
|
||||||
|
# All failed jobs in a build (--soft also includes soft-failed jobs;
|
||||||
|
# --all fetches every finished job):
|
||||||
|
.buildkite/scripts/ci-fetch-log.sh "https://buildkite.com/vllm/ci/builds/<N>"
|
||||||
|
|
||||||
|
# One job — `gh pr checks` URLs (#<job_uuid>) and web UI URLs (?sid=) both
|
||||||
|
# work; pass "-" as a second argument to stream to stdout:
|
||||||
|
.buildkite/scripts/ci-fetch-log.sh "https://buildkite.com/vllm/ci/builds/<N>#<job_uuid>"
|
||||||
|
```
|
||||||
|
|
||||||
|
To clean an already-downloaded log with `.buildkite/scripts/ci-clean-log.sh`:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./ci-clean-log.sh ci.log
|
||||||
|
```
|
||||||
|
|
||||||
|
## Reference
|
||||||
|
|
||||||
|
See [docs/contributing/ci/failures.md](../../../docs/contributing/ci/failures.md) for the full guide: filing CI failure issues, investigating/bisecting, reproducing flaky tests, and daily triage.
|
||||||
+13
-22
@@ -2,17 +2,16 @@
|
|||||||
# for more info about CODEOWNERS file
|
# for more info about CODEOWNERS file
|
||||||
|
|
||||||
# This lists cover the "core" components of vLLM that require careful review
|
# This lists cover the "core" components of vLLM that require careful review
|
||||||
/vllm/compilation @zou3519 @youkaichao @ProExpertProg @BoyuanFeng @vadiklyutiy
|
/vllm/compilation @zou3519 @youkaichao @ProExpertProg @BoyuanFeng
|
||||||
/vllm/distributed/kv_transfer @NickLucche @ApostaC @orozery @xuechendi
|
/vllm/distributed/kv_transfer @NickLucche @ApostaC @orozery @xuechendi @ivanium
|
||||||
/vllm/lora @jeejeelee
|
/vllm/lora @jeejeelee
|
||||||
/vllm/model_executor/layers/attention @LucasWilkinson @MatthewBonanni
|
/vllm/model_executor/layers/attention @LucasWilkinson @MatthewBonanni
|
||||||
/vllm/model_executor/layers/fused_moe @mgoin @pavanimajety @zyongye
|
/vllm/model_executor/layers/fused_moe @mgoin @pavanimajety @zyongye
|
||||||
/vllm/model_executor/layers/quantization @mgoin @robertgshaw2-redhat @tlrmchlsmth @yewentao256 @pavanimajety @zyongye
|
/vllm/model_executor/layers/quantization @mgoin @robertgshaw2-redhat @tlrmchlsmth @yewentao256 @pavanimajety @zyongye
|
||||||
/vllm/model_executor/layers/mamba @tdoublep @tomeras91
|
/vllm/model_executor/layers/mamba @tdoublep @tomeras91
|
||||||
/vllm/model_executor/layers/mamba/gdn_linear_attn.py @tdoublep @ZJY0516 @vadiklyutiy
|
/vllm/model_executor/layers/mamba/gdn/qwen_gdn_linear_attn.py @tdoublep @ZJY0516 @vadiklyutiy
|
||||||
/vllm/model_executor/layers/rotary_embedding.py @vadiklyutiy
|
|
||||||
/vllm/model_executor/model_loader @22quinn
|
/vllm/model_executor/model_loader @22quinn
|
||||||
/vllm/model_executor/layers/batch_invariant.py @yewentao256
|
/vllm/model_executor/layers/batch_invariant.py @yewentao256
|
||||||
/vllm/ir @ProExpertProg
|
/vllm/ir @ProExpertProg
|
||||||
/vllm/kernels/ @ProExpertProg @tjtanaa
|
/vllm/kernels/ @ProExpertProg @tjtanaa
|
||||||
/vllm/kernels/helion @ProExpertProg @zou3519
|
/vllm/kernels/helion @ProExpertProg @zou3519
|
||||||
@@ -24,7 +23,7 @@
|
|||||||
# Any change to the VllmConfig changes can have a large user-facing impact,
|
# Any change to the VllmConfig changes can have a large user-facing impact,
|
||||||
# so spam a lot of people
|
# so spam a lot of people
|
||||||
/vllm/config @WoosukKwon @youkaichao @robertgshaw2-redhat @mgoin @tlrmchlsmth @houseroad @yewentao256 @ProExpertProg
|
/vllm/config @WoosukKwon @youkaichao @robertgshaw2-redhat @mgoin @tlrmchlsmth @houseroad @yewentao256 @ProExpertProg
|
||||||
/vllm/config/cache.py @heheda12345
|
/vllm/config/cache.py @heheda12345 @ivanium
|
||||||
|
|
||||||
# Config utils
|
# Config utils
|
||||||
/vllm/config/utils.py @hmellor
|
/vllm/config/utils.py @hmellor
|
||||||
@@ -68,16 +67,17 @@
|
|||||||
/vllm/v1/attention/backends/flashinfer.py @mgoin @pavanimajety @vadiklyutiy
|
/vllm/v1/attention/backends/flashinfer.py @mgoin @pavanimajety @vadiklyutiy
|
||||||
/vllm/v1/attention/backends/triton_attn.py @tdoublep
|
/vllm/v1/attention/backends/triton_attn.py @tdoublep
|
||||||
/vllm/v1/attention/backends/gdn_attn.py @ZJY0516 @vadiklyutiy
|
/vllm/v1/attention/backends/gdn_attn.py @ZJY0516 @vadiklyutiy
|
||||||
/vllm/v1/core @WoosukKwon @robertgshaw2-redhat @njhill @ywang96 @alexm-redhat @heheda12345 @ApostaC @orozery
|
/vllm/v1/core @WoosukKwon @robertgshaw2-redhat @njhill @ywang96 @alexm-redhat @heheda12345 @ApostaC @orozery @ivanium
|
||||||
/vllm/v1/sample @22quinn @houseroad @njhill
|
/vllm/v1/sample @22quinn @houseroad @njhill
|
||||||
/vllm/v1/spec_decode @benchislett @luccafong @MatthewBonanni
|
/vllm/v1/spec_decode @benchislett @luccafong @MatthewBonanni
|
||||||
/vllm/v1/structured_output @mgoin @russellb @aarnphm @benchislett
|
/vllm/v1/structured_output @mgoin @russellb @aarnphm @benchislett
|
||||||
/vllm/v1/kv_cache_interface.py @heheda12345
|
/vllm/v1/kv_cache_interface.py @heheda12345 @ivanium
|
||||||
/vllm/v1/kv_offload @ApostaC @orozery
|
/vllm/v1/kv_offload @ApostaC @orozery
|
||||||
|
/vllm/v1/simple_kv_offload @ivanium
|
||||||
/vllm/v1/engine @njhill
|
/vllm/v1/engine @njhill
|
||||||
/vllm/v1/executor @njhill
|
/vllm/v1/executor @njhill
|
||||||
/vllm/v1/worker @njhill
|
/vllm/v1/worker @njhill
|
||||||
/vllm/v1/worker/kv_connector_model_runner_mixin.py @orozery @NickLucche
|
/vllm/v1/worker/kv_connector_model_runner_mixin.py @orozery @NickLucche @ivanium
|
||||||
|
|
||||||
# Model runner V2
|
# Model runner V2
|
||||||
/vllm/v1/worker/gpu @WoosukKwon @njhill @yewentao256
|
/vllm/v1/worker/gpu @WoosukKwon @njhill @yewentao256
|
||||||
@@ -104,13 +104,14 @@
|
|||||||
/tests/test_inputs.py @DarkLight1337 @ywang96
|
/tests/test_inputs.py @DarkLight1337 @ywang96
|
||||||
/tests/entrypoints/llm/test_struct_output_generate.py @mgoin @russellb @aarnphm
|
/tests/entrypoints/llm/test_struct_output_generate.py @mgoin @russellb @aarnphm
|
||||||
/tests/v1/structured_output @mgoin @russellb @aarnphm
|
/tests/v1/structured_output @mgoin @russellb @aarnphm
|
||||||
/tests/v1/core @WoosukKwon @robertgshaw2-redhat @njhill @ywang96 @alexm-redhat @heheda12345 @ApostaC @orozery
|
/tests/v1/core @WoosukKwon @robertgshaw2-redhat @njhill @ywang96 @alexm-redhat @heheda12345 @ApostaC @orozery @ivanium
|
||||||
/tests/weight_loading @mgoin @youkaichao @yewentao256
|
/tests/weight_loading @mgoin @youkaichao @yewentao256
|
||||||
/tests/lora @jeejeelee
|
/tests/lora @jeejeelee
|
||||||
/tests/models/language/generation/test_hybrid.py @tdoublep @tomeras91
|
/tests/models/language/generation/test_hybrid.py @tdoublep @tomeras91
|
||||||
/tests/v1/kv_connector/nixl_integration @NickLucche
|
/tests/v1/kv_connector/nixl_integration @NickLucche
|
||||||
/tests/v1/kv_connector @ApostaC @orozery
|
/tests/v1/kv_connector @ApostaC @orozery @ivanium
|
||||||
/tests/v1/kv_offload @ApostaC @orozery
|
/tests/v1/kv_offload @ApostaC @orozery
|
||||||
|
/tests/v1/simple_kv_offload @ivanium
|
||||||
/tests/v1/determinism @yewentao256
|
/tests/v1/determinism @yewentao256
|
||||||
/tests/reasoning @aarnphm @chaunceyjiang @sfeng33 @bbrowning
|
/tests/reasoning @aarnphm @chaunceyjiang @sfeng33 @bbrowning
|
||||||
/tests/tool_parsers @aarnphm @chaunceyjiang @sfeng33 @bbrowning
|
/tests/tool_parsers @aarnphm @chaunceyjiang @sfeng33 @bbrowning
|
||||||
@@ -118,17 +119,7 @@
|
|||||||
|
|
||||||
# Transformers modeling backend
|
# Transformers modeling backend
|
||||||
/vllm/model_executor/models/transformers @hmellor
|
/vllm/model_executor/models/transformers @hmellor
|
||||||
/tests/models/test_transformers.py @hmellor
|
/tests/models/transformers @hmellor
|
||||||
|
|
||||||
# Observability
|
|
||||||
/vllm/config/observability.py @markmc
|
|
||||||
/vllm/v1/metrics @markmc
|
|
||||||
/tests/v1/metrics @markmc
|
|
||||||
/vllm/tracing.py @markmc
|
|
||||||
/tests/v1/tracing/test_tracing.py @markmc
|
|
||||||
/vllm/config/kv_events.py @markmc
|
|
||||||
/vllm/distributed/kv_events.py @markmc
|
|
||||||
/tests/distributed/test_events.py @markmc
|
|
||||||
|
|
||||||
# Docs
|
# Docs
|
||||||
/docs/mkdocs @hmellor
|
/docs/mkdocs @hmellor
|
||||||
|
|||||||
@@ -0,0 +1,7 @@
|
|||||||
|
# Custom self-hosted runner labels (e.g. the autoscaling vllm-runners pool) so
|
||||||
|
# actionlint doesn't flag them as unknown in `runs-on`.
|
||||||
|
self-hosted-runner:
|
||||||
|
labels:
|
||||||
|
- vllm-runners
|
||||||
|
# Not yet in actionlint's known-label set.
|
||||||
|
- macos-26
|
||||||
@@ -388,9 +388,13 @@ pull_request_rules:
|
|||||||
- or:
|
- or:
|
||||||
- files~=^tests/tool_use/
|
- files~=^tests/tool_use/
|
||||||
- files~=^tests/tool_parsers/
|
- files~=^tests/tool_parsers/
|
||||||
|
- files~=^tests/parser/
|
||||||
|
- files~=^tests/reasoning/
|
||||||
- files~=^tests/entrypoints/openai/.*tool.*
|
- files~=^tests/entrypoints/openai/.*tool.*
|
||||||
- files~=^tests/entrypoints/anthropic/.*tool.*
|
- files~=^tests/entrypoints/anthropic/.*tool.*
|
||||||
- files~=^vllm/tool_parsers/
|
- files~=^vllm/tool_parsers/
|
||||||
|
- files~=^vllm/parser/
|
||||||
|
- files~=^vllm/reasoning/
|
||||||
- files=docs/features/tool_calling.md
|
- files=docs/features/tool_calling.md
|
||||||
- files~=^examples/tool_calling/
|
- files~=^examples/tool_calling/
|
||||||
actions:
|
actions:
|
||||||
|
|||||||
@@ -327,7 +327,7 @@ jobs:
|
|||||||
message: 'CC {users} for ROCm-related issue',
|
message: 'CC {users} for ROCm-related issue',
|
||||||
},
|
},
|
||||||
mistral: {
|
mistral: {
|
||||||
users: ['patrickvonplaten', 'juliendenize', 'andylolu2'],
|
users: ['patrickvonplaten', 'juliendenize', 'andylolu2', 'NickLucche'],
|
||||||
message: 'CC {users} for Mistral-related issue',
|
message: 'CC {users} for Mistral-related issue',
|
||||||
},
|
},
|
||||||
// Add more label -> user mappings here
|
// Add more label -> user mappings here
|
||||||
|
|||||||
@@ -11,13 +11,25 @@ permissions:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
macos-m1-smoke-test:
|
macos-m1-smoke-test:
|
||||||
runs-on: macos-latest
|
# macos-26 (the supported target) is still a preview runner, so gate on GA
|
||||||
|
# macos-15 and keep macos-26 non-blocking.
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
include:
|
||||||
|
- os: macos-15
|
||||||
|
required: true
|
||||||
|
- os: macos-26
|
||||||
|
required: false
|
||||||
|
name: macos-m1-smoke-test (${{ matrix.os }})
|
||||||
|
runs-on: ${{ matrix.os }}
|
||||||
|
continue-on-error: ${{ !matrix.required }}
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v6.0.1
|
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||||
|
|
||||||
- uses: astral-sh/setup-uv@v7
|
- uses: astral-sh/setup-uv@37802adc94f370d6bfd71619e3f0bf239e1f3b78 # v7.6.0
|
||||||
with:
|
with:
|
||||||
enable-cache: true
|
enable-cache: true
|
||||||
cache-dependency-glob: |
|
cache-dependency-glob: |
|
||||||
@@ -72,14 +84,11 @@ jobs:
|
|||||||
# Test health endpoint
|
# Test health endpoint
|
||||||
curl -f http://localhost:8000/health
|
curl -f http://localhost:8000/health
|
||||||
|
|
||||||
# Test completion
|
# Long prompt: hits the split-KV path that short prompts skip (#46769).
|
||||||
curl -f http://localhost:8000/v1/completions \
|
PAYLOAD=$(python -c "import json; print(json.dumps({'model': 'Qwen/Qwen3-0.6B', 'prompt': 'The quick brown fox jumps over the lazy dog. ' * 24, 'max_tokens': 16}))")
|
||||||
|
curl -f --max-time 120 http://localhost:8000/v1/completions \
|
||||||
-H "Content-Type: application/json" \
|
-H "Content-Type: application/json" \
|
||||||
-d '{
|
-d "$PAYLOAD"
|
||||||
"model": "Qwen/Qwen3-0.6B",
|
|
||||||
"prompt": "Hello",
|
|
||||||
"max_tokens": 5
|
|
||||||
}'
|
|
||||||
|
|
||||||
# Cleanup
|
# Cleanup
|
||||||
kill "$SERVER_PID"
|
kill "$SERVER_PID"
|
||||||
|
|||||||
@@ -28,7 +28,8 @@ jobs:
|
|||||||
pull_number: context.payload.pull_request.number,
|
pull_number: context.payload.pull_request.number,
|
||||||
});
|
});
|
||||||
|
|
||||||
const hasReadyLabel = pr.labels.some(l => l.name === 'ready');
|
const readyLabels = ['ready', 'ready-run-all-tests'];
|
||||||
|
const hasReadyLabel = pr.labels.some(l => readyLabels.includes(l.name));
|
||||||
const hasVerifiedLabel = pr.labels.some(l => l.name === 'verified');
|
const hasVerifiedLabel = pr.labels.some(l => l.name === 'verified');
|
||||||
|
|
||||||
const { data: mergedPRs } = await github.rest.search.issuesAndPullRequests({
|
const { data: mergedPRs } = await github.rest.search.issuesAndPullRequests({
|
||||||
@@ -40,18 +41,22 @@ jobs:
|
|||||||
if (hasReadyLabel || hasVerifiedLabel || mergedCount >= 4) {
|
if (hasReadyLabel || hasVerifiedLabel || mergedCount >= 4) {
|
||||||
core.info(`Check passed: verified label=${hasVerifiedLabel}, ready label=${hasReadyLabel}, 4+ merged PRs=${mergedCount >= 4}`);
|
core.info(`Check passed: verified label=${hasVerifiedLabel}, ready label=${hasReadyLabel}, 4+ merged PRs=${mergedCount >= 4}`);
|
||||||
} else {
|
} else {
|
||||||
core.setFailed(`PR must have the 'verified' or 'ready' (which also triggers tests) label or the author must have at least 4 merged PRs (found ${mergedCount}).`);
|
core.setFailed(`PR must have the 'verified', 'ready', or 'ready-run-all-tests' label (the ready labels also trigger tests) or the author must have at least 4 merged PRs (found ${mergedCount}).`);
|
||||||
}
|
}
|
||||||
|
|
||||||
pre-commit:
|
pre-commit:
|
||||||
needs: pre-run-check
|
needs: pre-run-check
|
||||||
if: always() && (needs.pre-run-check.result == 'success' || needs.pre-run-check.result == 'skipped')
|
if: always() && (needs.pre-run-check.result == 'success' || needs.pre-run-check.result == 'skipped')
|
||||||
runs-on: ubuntu-latest
|
runs-on: [self-hosted, linux, x64, vllm-runners]
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||||
- uses: actions/setup-python@83679a892e2d95755f2dac6acb0bfd1e9ac5d548 # v6.1.0
|
- uses: actions/setup-python@83679a892e2d95755f2dac6acb0bfd1e9ac5d548 # v6.1.0
|
||||||
with:
|
with:
|
||||||
python-version: "3.12"
|
python-version: "3.12"
|
||||||
|
# Provide shellcheck on PATH so tools/pre_commit/shellcheck.sh skips its
|
||||||
|
# wget + tar -xJ self-download, which the self-hosted runner image lacks
|
||||||
|
# (no wget/xz). Pinned to shellcheck 0.10.0 to match the script's "stable".
|
||||||
|
- run: python -m pip install shellcheck-py==0.10.0.1
|
||||||
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
|
- run: echo "::add-matcher::.github/workflows/matchers/actionlint.json"
|
||||||
- run: echo "::add-matcher::.github/workflows/matchers/markdownlint.json"
|
- run: echo "::add-matcher::.github/workflows/matchers/markdownlint.json"
|
||||||
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
|
- run: echo "::add-matcher::.github/workflows/matchers/mypy.json"
|
||||||
|
|||||||
+3
-1
@@ -199,7 +199,9 @@ cython_debug/
|
|||||||
.vscode/
|
.vscode/
|
||||||
|
|
||||||
# Claude
|
# Claude
|
||||||
.claude/
|
.claude/*
|
||||||
|
!.claude/skills/
|
||||||
|
!.claude/skills/**
|
||||||
|
|
||||||
# Codex
|
# Codex
|
||||||
.codex/
|
.codex/
|
||||||
|
|||||||
@@ -131,6 +131,19 @@ repos:
|
|||||||
--python-version, "3.12",
|
--python-version, "3.12",
|
||||||
]
|
]
|
||||||
files: ^requirements/(common|xpu|test/xpu)\.(in|txt)$
|
files: ^requirements/(common|xpu|test/xpu)\.(in|txt)$
|
||||||
|
- id: pip-compile
|
||||||
|
alias: pip-compile-cpu
|
||||||
|
name: pip-compile-cpu
|
||||||
|
args: [
|
||||||
|
requirements/test/cuda.in,
|
||||||
|
-o, requirements/test/cpu.txt,
|
||||||
|
--index-strategy, unsafe-best-match,
|
||||||
|
--torch-backend, cpu,
|
||||||
|
--python-platform, x86_64-manylinux_2_28,
|
||||||
|
--python-version, "3.12",
|
||||||
|
]
|
||||||
|
files: ^requirements/(common|cpu|test/(cuda|cpu))\.(in|txt)$
|
||||||
|
exclude: ^requirements/test/cuda\.txt$
|
||||||
- id: pip-compile
|
- id: pip-compile
|
||||||
alias: pip-compile-docs
|
alias: pip-compile-docs
|
||||||
name: pip-compile-docs
|
name: pip-compile-docs
|
||||||
|
|||||||
@@ -29,6 +29,7 @@ Do not open one-off PRs for tiny edits (single typo, isolated style change, one
|
|||||||
- PR descriptions for AI-assisted work **must** include:
|
- PR descriptions for AI-assisted work **must** include:
|
||||||
- Why this is not duplicating an existing PR.
|
- Why this is not duplicating an existing PR.
|
||||||
- Test commands run and results.
|
- Test commands run and results.
|
||||||
|
- Model evaluation results when the change affects output, accuracy, or serving.
|
||||||
- Clear statement that AI assistance was used.
|
- Clear statement that AI assistance was used.
|
||||||
|
|
||||||
### Fail-closed behavior
|
### Fail-closed behavior
|
||||||
@@ -66,23 +67,38 @@ VLLM_USE_PRECOMPILED=1 uv pip install -e . --torch-backend=auto
|
|||||||
uv pip install -e . --torch-backend=auto
|
uv pip install -e . --torch-backend=auto
|
||||||
```
|
```
|
||||||
|
|
||||||
### Running tests
|
### Tests
|
||||||
|
|
||||||
> Requires [Environment setup](#environment-setup) and [Installing dependencies](#installing-dependencies).
|
> Requires [Environment setup](#environment-setup) and [Installing dependencies](#installing-dependencies).
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Install test dependencies.
|
# Install test dependencies (use cuda.in on non-x86_64):
|
||||||
# requirements/test/cuda.txt is pinned to x86_64; on other platforms, use the
|
uv pip install -r requirements/test/cuda.in
|
||||||
# unpinned source file instead:
|
|
||||||
uv pip install -r requirements/test/cuda.in # resolves for current platform
|
|
||||||
# Or on x86_64:
|
|
||||||
uv pip install -r requirements/test/cuda.txt
|
|
||||||
|
|
||||||
# Run a specific test file (use .venv/bin/python directly;
|
# Run a specific test file:
|
||||||
# `source activate` does not persist in non-interactive shells):
|
|
||||||
.venv/bin/python -m pytest tests/path/to/test_file.py -v
|
.venv/bin/python -m pytest tests/path/to/test_file.py -v
|
||||||
```
|
```
|
||||||
|
|
||||||
|
When adding tests:
|
||||||
|
|
||||||
|
- **Design before you write.** Answer four questions first: what is the module
|
||||||
|
for, what is its I/O contract, what failure am I guarding against, and what is
|
||||||
|
the cheapest level that catches it (unit over integration over e2e)?
|
||||||
|
- **Reuse before create.** Extend existing test files, `conftest.py` fixtures, and
|
||||||
|
helpers; add a new file only when no nearby suite fits.
|
||||||
|
- **Test behavior with intent.** Assert observable outcomes through public APIs;
|
||||||
|
state why in the name or docstring. Skip trivial wiring; flaky tests are worse
|
||||||
|
than no tests.
|
||||||
|
- **Keep it minimal.** One behavior per test and the smallest setup that
|
||||||
|
triggers it; if the test diff dwarfs the code change, cut scope.
|
||||||
|
- **No one-off kernel benchmarks in `tests/`.** Put kernel perf work in
|
||||||
|
`benchmarks/kernels/`; prove correctness in existing pytest suites.
|
||||||
|
- **Run model evals for model-affecting changes.** Search `tests/evals/` or use
|
||||||
|
`vllm bench` and include results in the PR — do not wait for reviewers to ask.
|
||||||
|
|
||||||
|
For model-specific requirements, see
|
||||||
|
[`docs/contributing/model/tests.md`](docs/contributing/model/tests.md).
|
||||||
|
|
||||||
### Running linters
|
### Running linters
|
||||||
|
|
||||||
> Requires [Environment setup](#environment-setup).
|
> Requires [Environment setup](#environment-setup).
|
||||||
@@ -107,34 +123,18 @@ Use [Google-style docstrings](https://google.github.io/styleguide/pyguide.html#3
|
|||||||
|
|
||||||
### Coding style guidelines
|
### Coding style guidelines
|
||||||
|
|
||||||
Follow these rules for all code changes in this repository:
|
- Match existing code style
|
||||||
|
- Minimize use of comments. Eliminate comments which are redundant, preferring legible and self-documenting code. When used, keep docstrings and comments brief and direct.
|
||||||
- Try to match existing code style.
|
|
||||||
- Code should be self-documenting and self-explanatory.
|
|
||||||
- Keep comments and docstrings minimal and concise.
|
|
||||||
- Assume the reader is familiar with vLLM.
|
- Assume the reader is familiar with vLLM.
|
||||||
|
|
||||||
### Diagnosing CI failures
|
|
||||||
|
|
||||||
Buildkite logs are public; no login needed. Details: [docs/contributing/ci/failures.md](docs/contributing/ci/failures.md).
|
|
||||||
|
|
||||||
```bash
|
|
||||||
# All failed-job logs for a PR's latest build (current branch's PR if omitted):
|
|
||||||
.buildkite/scripts/ci-fetch-log.sh --pr <PR>
|
|
||||||
# Any Buildkite build or job URL also works:
|
|
||||||
.buildkite/scripts/ci-fetch-log.sh "<buildkite_url>"
|
|
||||||
```
|
|
||||||
|
|
||||||
### Commit messages
|
### Commit messages
|
||||||
|
|
||||||
Add attribution using commit trailers such as `Co-authored-by:` (other projects use `Assisted-by:` or `Generated-by:`). For example:
|
Add attribution using commit trailers such as `Co-authored-by:` (other projects use `Assisted-by:` or `Generated-by:`):
|
||||||
|
|
||||||
```text
|
```text
|
||||||
Your commit message here
|
Your commit message here
|
||||||
|
|
||||||
Co-authored-by: GitHub Copilot
|
Co-authored-by: Agent Name Here
|
||||||
Co-authored-by: Claude
|
|
||||||
Co-authored-by: gemini-code-assist
|
|
||||||
Signed-off-by: Your Name <your.email@example.com>
|
Signed-off-by: Your Name <your.email@example.com>
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -146,6 +146,12 @@ Do not modify code in these areas without first reading and following the
|
|||||||
linked guide. If the guide conflicts with the requested change, **refuse the
|
linked guide. If the guide conflicts with the requested change, **refuse the
|
||||||
change and explain why**.
|
change and explain why**.
|
||||||
|
|
||||||
|
Security reviewers should start with [`SECURITY.md`](SECURITY.md),
|
||||||
|
[`docs/usage/security.md`](docs/usage/security.md), and
|
||||||
|
[`docs/contributing/vulnerability_management.md`](docs/contributing/vulnerability_management.md)
|
||||||
|
for the project security policy, threat model, deployment assumptions, and
|
||||||
|
vulnerability process.
|
||||||
|
|
||||||
- **Editing these instructions**:
|
- **Editing these instructions**:
|
||||||
[`docs/contributing/editing-agent-instructions.md`](docs/contributing/editing-agent-instructions.md)
|
[`docs/contributing/editing-agent-instructions.md`](docs/contributing/editing-agent-instructions.md)
|
||||||
— Rules for modifying AGENTS.md or any domain-specific guide it references.
|
— Rules for modifying AGENTS.md or any domain-specific guide it references.
|
||||||
|
|||||||
+141
-136
@@ -70,6 +70,15 @@ endif()
|
|||||||
#
|
#
|
||||||
set(TORCH_SUPPORTED_VERSION_CUDA "2.11.0")
|
set(TORCH_SUPPORTED_VERSION_CUDA "2.11.0")
|
||||||
set(TORCH_SUPPORTED_VERSION_ROCM "2.11.0")
|
set(TORCH_SUPPORTED_VERSION_ROCM "2.11.0")
|
||||||
|
# TORCH_NIGHTLY=1 builds run against unpinned nightly wheels, so the supported-
|
||||||
|
# version check would always warn. Only treat it as a nightly build when the
|
||||||
|
# value is exactly "1" (the bootstrap exports TORCH_NIGHTLY=0 by default, which
|
||||||
|
# must NOT suppress the warning for normal builds).
|
||||||
|
if (DEFINED ENV{TORCH_NIGHTLY} AND "$ENV{TORCH_NIGHTLY}" STREQUAL "1")
|
||||||
|
set(TORCH_NIGHTLY_BUILD TRUE)
|
||||||
|
else()
|
||||||
|
set(TORCH_NIGHTLY_BUILD FALSE)
|
||||||
|
endif()
|
||||||
|
|
||||||
#
|
#
|
||||||
# Try to find python package with an executable that exactly matches
|
# Try to find python package with an executable that exactly matches
|
||||||
@@ -140,6 +149,21 @@ if(Python_VERSION VERSION_GREATER_EQUAL "3.11")
|
|||||||
WITH_SOABI)
|
WITH_SOABI)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
#
|
||||||
|
# fs_io extension (pure CXX; must stay above the non-CUDA device branch
|
||||||
|
# so CPU builds define the target before the early return).
|
||||||
|
# GIL-releasing filesystem helpers for FileSystemTierManager.
|
||||||
|
#
|
||||||
|
if(Python_VERSION VERSION_GREATER_EQUAL "3.11")
|
||||||
|
define_extension_target(
|
||||||
|
fs_io_C
|
||||||
|
DESTINATION vllm
|
||||||
|
LANGUAGE CXX
|
||||||
|
SOURCES csrc/fs_io.cpp
|
||||||
|
USE_SABI 3.11
|
||||||
|
WITH_SOABI)
|
||||||
|
endif()
|
||||||
|
|
||||||
#
|
#
|
||||||
# Forward the non-CUDA device extensions to external CMake scripts.
|
# Forward the non-CUDA device extensions to external CMake scripts.
|
||||||
#
|
#
|
||||||
@@ -160,7 +184,7 @@ endif()
|
|||||||
if (NOT HIP_FOUND AND NOT PYTORCH_FOUND_HIP AND CUDA_FOUND)
|
if (NOT HIP_FOUND AND NOT PYTORCH_FOUND_HIP AND CUDA_FOUND)
|
||||||
set(VLLM_GPU_LANG "CUDA")
|
set(VLLM_GPU_LANG "CUDA")
|
||||||
|
|
||||||
if (NOT Torch_VERSION VERSION_EQUAL ${TORCH_SUPPORTED_VERSION_CUDA})
|
if (NOT TORCH_NIGHTLY_BUILD AND NOT Torch_VERSION VERSION_EQUAL ${TORCH_SUPPORTED_VERSION_CUDA})
|
||||||
message(WARNING "Pytorch version ${TORCH_SUPPORTED_VERSION_CUDA} "
|
message(WARNING "Pytorch version ${TORCH_SUPPORTED_VERSION_CUDA} "
|
||||||
"expected for CUDA build, saw ${Torch_VERSION} instead.")
|
"expected for CUDA build, saw ${Torch_VERSION} instead.")
|
||||||
endif()
|
endif()
|
||||||
@@ -173,7 +197,7 @@ elseif(HIP_FOUND OR PYTORCH_FOUND_HIP)
|
|||||||
enable_language(HIP)
|
enable_language(HIP)
|
||||||
|
|
||||||
# ROCm 5.X and 6.X
|
# ROCm 5.X and 6.X
|
||||||
if (ROCM_VERSION_DEV_MAJOR GREATER_EQUAL 5 AND
|
if (NOT TORCH_NIGHTLY_BUILD AND ROCM_VERSION_DEV_MAJOR GREATER_EQUAL 5 AND
|
||||||
Torch_VERSION VERSION_LESS ${TORCH_SUPPORTED_VERSION_ROCM})
|
Torch_VERSION VERSION_LESS ${TORCH_SUPPORTED_VERSION_ROCM})
|
||||||
message(WARNING "Pytorch version >= ${TORCH_SUPPORTED_VERSION_ROCM} "
|
message(WARNING "Pytorch version >= ${TORCH_SUPPORTED_VERSION_ROCM} "
|
||||||
"expected for ROCm build, saw ${Torch_VERSION} instead.")
|
"expected for ROCm build, saw ${Torch_VERSION} instead.")
|
||||||
@@ -270,6 +294,16 @@ if(VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
#
|
#
|
||||||
set(CMAKE_${VLLM_GPU_LANG}_FLAGS "${CMAKE_${VLLM_GPU_LANG}_FLAGS} -Wno-unused-result -Wno-unused-value")
|
set(CMAKE_${VLLM_GPU_LANG}_FLAGS "${CMAKE_${VLLM_GPU_LANG}_FLAGS} -Wno-unused-result -Wno-unused-value")
|
||||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Wno-unused-result -Wno-unused-value")
|
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Wno-unused-result -Wno-unused-value")
|
||||||
|
|
||||||
|
# When using LTO then *.cpp files must be compiled with same compiler as used linker
|
||||||
|
# So if HIP uses clang linker we also must use it
|
||||||
|
# Otherwise symbols will be missing from .so
|
||||||
|
if (CMAKE_CXX_FLAGS MATCHES "\-flto")
|
||||||
|
if(NOT CMAKE_CXX_COMPILER_ID STREQUAL CMAKE_HIP_COMPILER_ID)
|
||||||
|
message(FATAL_ERROR "LTO is enabled for ROCm build, but the C++ compiler (${CMAKE_CXX_COMPILER_ID}) and HIP compiler (${CMAKE_HIP_COMPILER_ID}) are different which is not supported. "
|
||||||
|
"Please ensure they are same by setting CXX=${CMAKE_HIP_COMPILER} environment variable. Or alternatively disable LTO.")
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
#
|
#
|
||||||
@@ -319,111 +353,33 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
endif()
|
endif()
|
||||||
|
|
||||||
#
|
#
|
||||||
# _C extension
|
# Legacy _C extension (ROCm only — CUDA ops migrated to _C_stable_libtorch)
|
||||||
#
|
#
|
||||||
|
|
||||||
set(VLLM_EXT_SRC
|
if(VLLM_GPU_LANG STREQUAL "HIP")
|
||||||
"csrc/quantization/activation_kernels.cu"
|
set(VLLM_EXT_SRC
|
||||||
"csrc/torch_bindings.cpp")
|
"csrc/torch_bindings.cpp"
|
||||||
|
"csrc/custom_quickreduce.cu")
|
||||||
|
|
||||||
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
message(STATUS "Enabling C extension.")
|
||||||
SET(CUTLASS_ENABLE_HEADERS_ONLY ON CACHE BOOL "Enable only the header library")
|
define_extension_target(
|
||||||
|
_C
|
||||||
|
DESTINATION vllm
|
||||||
|
LANGUAGE ${VLLM_GPU_LANG}
|
||||||
|
SOURCES ${VLLM_EXT_SRC}
|
||||||
|
COMPILE_FLAGS ${VLLM_GPU_FLAGS}
|
||||||
|
ARCHITECTURES ${VLLM_GPU_ARCHES}
|
||||||
|
INCLUDE_DIRECTORIES ${CUTLASS_INCLUDE_DIR}
|
||||||
|
INCLUDE_DIRECTORIES ${CUTLASS_TOOLS_UTIL_INCLUDE_DIR}
|
||||||
|
USE_SABI 3
|
||||||
|
WITH_SOABI)
|
||||||
|
|
||||||
# Set CUTLASS_REVISION. Used for FetchContent. Also fixes some bogus messages when building.
|
# If CUTLASS is compiled on NVCC >= 12.5, it by default uses
|
||||||
set(CUTLASS_REVISION "v4.4.2")
|
# cudaGetDriverEntryPointByVersion as a wrapper to avoid directly calling the
|
||||||
|
# driver API. This causes problems when linking with earlier versions of CUDA.
|
||||||
# Use the specified CUTLASS source directory for compilation if VLLM_CUTLASS_SRC_DIR is provided
|
# Setting this variable sidesteps the issue by calling the driver directly.
|
||||||
if (DEFINED ENV{VLLM_CUTLASS_SRC_DIR})
|
target_compile_definitions(_C PRIVATE CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
||||||
set(VLLM_CUTLASS_SRC_DIR $ENV{VLLM_CUTLASS_SRC_DIR})
|
endif() # _C HIP endif
|
||||||
endif()
|
|
||||||
|
|
||||||
if(VLLM_CUTLASS_SRC_DIR)
|
|
||||||
if(NOT IS_ABSOLUTE VLLM_CUTLASS_SRC_DIR)
|
|
||||||
get_filename_component(VLLM_CUTLASS_SRC_DIR "${VLLM_CUTLASS_SRC_DIR}" ABSOLUTE)
|
|
||||||
endif()
|
|
||||||
message(STATUS "The VLLM_CUTLASS_SRC_DIR is set, using ${VLLM_CUTLASS_SRC_DIR} for compilation")
|
|
||||||
FetchContent_Declare(cutlass SOURCE_DIR ${VLLM_CUTLASS_SRC_DIR})
|
|
||||||
else()
|
|
||||||
FetchContent_Declare(
|
|
||||||
cutlass
|
|
||||||
GIT_REPOSITORY https://github.com/nvidia/cutlass.git
|
|
||||||
# Please keep this in sync with CUTLASS_REVISION line above.
|
|
||||||
GIT_TAG ${CUTLASS_REVISION}
|
|
||||||
GIT_PROGRESS TRUE
|
|
||||||
|
|
||||||
# Speed up CUTLASS download by retrieving only the specified GIT_TAG instead of the history.
|
|
||||||
# Important: If GIT_SHALLOW is enabled then GIT_TAG works only with branch names and tags.
|
|
||||||
# So if the GIT_TAG above is updated to a commit hash, GIT_SHALLOW must be set to FALSE
|
|
||||||
GIT_SHALLOW TRUE
|
|
||||||
)
|
|
||||||
endif()
|
|
||||||
FetchContent_MakeAvailable(cutlass)
|
|
||||||
|
|
||||||
set_gencode_flags_for_srcs(
|
|
||||||
SRCS "${VLLM_EXT_SRC}"
|
|
||||||
CUDA_ARCHS "${CUDA_ARCHS}")
|
|
||||||
|
|
||||||
# Expert-specialization MXFP8 blockscaled grouped kernels (SM100+).
|
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
|
||||||
cuda_archs_loose_intersection(ES_MXFP8_GROUPED_MM_ARCHS "10.0f;11.0f" "${CUDA_ARCHS}")
|
|
||||||
else()
|
|
||||||
cuda_archs_loose_intersection(ES_MXFP8_GROUPED_MM_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
|
||||||
endif()
|
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8 AND ES_MXFP8_GROUPED_MM_ARCHS)
|
|
||||||
set(ES_MXFP8_GROUPED_MM_SRCS
|
|
||||||
"csrc/libtorch_stable/moe/mxfp8_moe/cutlass_mxfp8_grouped_mm.cu"
|
|
||||||
"csrc/libtorch_stable/moe/mxfp8_moe/mxfp8_experts_quant.cu")
|
|
||||||
set_gencode_flags_for_srcs(
|
|
||||||
SRCS "${ES_MXFP8_GROUPED_MM_SRCS}"
|
|
||||||
CUDA_ARCHS "${ES_MXFP8_GROUPED_MM_ARCHS}")
|
|
||||||
list(APPEND VLLM_STABLE_EXT_SRC "${ES_MXFP8_GROUPED_MM_SRCS}")
|
|
||||||
list(APPEND VLLM_GPU_FLAGS "-DENABLE_ES_MXFP8_GROUPED_MM_SM100=1")
|
|
||||||
message(STATUS "Building ES MXFP8 grouped kernels for archs: ${ES_MXFP8_GROUPED_MM_ARCHS}")
|
|
||||||
else()
|
|
||||||
if (NOT ${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8
|
|
||||||
AND ES_MXFP8_GROUPED_MM_ARCHS)
|
|
||||||
message(STATUS "Not building ES MXFP8 grouped kernels as CUDA Compiler version is "
|
|
||||||
"not >= 12.8.")
|
|
||||||
else()
|
|
||||||
message(STATUS "Not building ES MXFP8 grouped kernels as no compatible archs found "
|
|
||||||
"in CUDA target architectures.")
|
|
||||||
endif()
|
|
||||||
endif()
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
# if CUDA endif
|
|
||||||
endif()
|
|
||||||
|
|
||||||
if (VLLM_GPU_LANG STREQUAL "HIP")
|
|
||||||
# Add QuickReduce kernels (ROCm-only; not part of stable ABI migration).
|
|
||||||
# TODO: Remove the cuda_view when ROCm upgrade to torch 2.11.
|
|
||||||
list(APPEND VLLM_EXT_SRC
|
|
||||||
"csrc/custom_quickreduce.cu"
|
|
||||||
"csrc/cuda_view.cu"
|
|
||||||
"csrc/libtorch_stable/cuda_utils_kernels.cu"
|
|
||||||
)
|
|
||||||
# if ROCM endif
|
|
||||||
endif()
|
|
||||||
|
|
||||||
message(STATUS "Enabling C extension.")
|
|
||||||
define_extension_target(
|
|
||||||
_C
|
|
||||||
DESTINATION vllm
|
|
||||||
LANGUAGE ${VLLM_GPU_LANG}
|
|
||||||
SOURCES ${VLLM_EXT_SRC}
|
|
||||||
COMPILE_FLAGS ${VLLM_GPU_FLAGS}
|
|
||||||
ARCHITECTURES ${VLLM_GPU_ARCHES}
|
|
||||||
INCLUDE_DIRECTORIES ${CUTLASS_INCLUDE_DIR}
|
|
||||||
INCLUDE_DIRECTORIES ${CUTLASS_TOOLS_UTIL_INCLUDE_DIR}
|
|
||||||
USE_SABI 3
|
|
||||||
WITH_SOABI)
|
|
||||||
|
|
||||||
# If CUTLASS is compiled on NVCC >= 12.5, it by default uses
|
|
||||||
# cudaGetDriverEntryPointByVersion as a wrapper to avoid directly calling the
|
|
||||||
# driver API. This causes problems when linking with earlier versions of CUDA.
|
|
||||||
# Setting this variable sidesteps the issue by calling the driver directly.
|
|
||||||
target_compile_definitions(_C PRIVATE CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
|
||||||
|
|
||||||
if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
||||||
#
|
#
|
||||||
@@ -431,7 +387,11 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
#
|
#
|
||||||
set(VLLM_STABLE_EXT_SRC
|
set(VLLM_STABLE_EXT_SRC
|
||||||
"csrc/libtorch_stable/torch_bindings.cpp"
|
"csrc/libtorch_stable/torch_bindings.cpp"
|
||||||
|
"csrc/libtorch_stable/cuda_view.cu"
|
||||||
|
"csrc/libtorch_stable/cuda_utils_kernels.cu"
|
||||||
"csrc/libtorch_stable/activation_kernels.cu"
|
"csrc/libtorch_stable/activation_kernels.cu"
|
||||||
|
"csrc/libtorch_stable/ngram_embedding_kernels.cu"
|
||||||
|
"csrc/libtorch_stable/quantization/activation_kernels.cu"
|
||||||
"csrc/libtorch_stable/quantization/w8a8/int8/scaled_quant.cu"
|
"csrc/libtorch_stable/quantization/w8a8/int8/scaled_quant.cu"
|
||||||
"csrc/libtorch_stable/quantization/w8a8/fp8/common.cu"
|
"csrc/libtorch_stable/quantization/w8a8/fp8/common.cu"
|
||||||
"csrc/libtorch_stable/quantization/w8a8/fp8/per_token_group_quant.cu"
|
"csrc/libtorch_stable/quantization/w8a8/fp8/per_token_group_quant.cu"
|
||||||
@@ -449,18 +409,63 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
"csrc/libtorch_stable/sampler.cu"
|
"csrc/libtorch_stable/sampler.cu"
|
||||||
"csrc/libtorch_stable/topk.cu"
|
"csrc/libtorch_stable/topk.cu"
|
||||||
"csrc/libtorch_stable/mamba/selective_scan_fwd.cu"
|
"csrc/libtorch_stable/mamba/selective_scan_fwd.cu"
|
||||||
"csrc/libtorch_stable/attention/paged_attention_v1.cu"
|
|
||||||
"csrc/libtorch_stable/attention/paged_attention_v2.cu"
|
|
||||||
"csrc/libtorch_stable/cache_kernels.cu"
|
|
||||||
"csrc/libtorch_stable/cache_kernels.cu"
|
"csrc/libtorch_stable/cache_kernels.cu"
|
||||||
"csrc/libtorch_stable/cache_kernels_fused.cu"
|
"csrc/libtorch_stable/cache_kernels_fused.cu"
|
||||||
"csrc/libtorch_stable/custom_all_reduce.cu"
|
"csrc/libtorch_stable/custom_all_reduce.cu"
|
||||||
"csrc/libtorch_stable/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu")
|
"csrc/libtorch_stable/fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu")
|
||||||
|
|
||||||
|
if(VLLM_GPU_LANG STREQUAL "CUDA" AND
|
||||||
|
DEFINED CMAKE_CUDA_COMPILER_VERSION AND
|
||||||
|
CMAKE_CUDA_COMPILER_VERSION VERSION_GREATER_EQUAL 12.0)
|
||||||
|
|
||||||
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
|
cuda_archs_loose_intersection(COOPERATIVE_TOPK_ARCHS
|
||||||
|
"9.0a;10.0f;10.1f;10.3f;11.0f;12.0f;12.1f" "${CUDA_ARCHS}")
|
||||||
|
else()
|
||||||
|
cuda_archs_loose_intersection(COOPERATIVE_TOPK_ARCHS
|
||||||
|
"9.0a;10.0a;10.1a;10.3a;12.0a;12.1a" "${CUDA_ARCHS}")
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(COOPERATIVE_TOPK_ARCHS)
|
||||||
|
list(APPEND VLLM_GPU_FLAGS "-DVLLM_ENABLE_COOPERATIVE_TOPK=1")
|
||||||
|
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
|
|
||||||
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
||||||
|
SET(CUTLASS_ENABLE_HEADERS_ONLY ON CACHE BOOL "Enable only the header library")
|
||||||
|
|
||||||
|
# Set CUTLASS_REVISION. Used for FetchContent. Also fixes some bogus messages when building.
|
||||||
|
set(CUTLASS_REVISION "v4.4.2")
|
||||||
|
|
||||||
|
# Use the specified CUTLASS source directory for compilation if VLLM_CUTLASS_SRC_DIR is provided
|
||||||
|
if (DEFINED ENV{VLLM_CUTLASS_SRC_DIR})
|
||||||
|
set(VLLM_CUTLASS_SRC_DIR $ENV{VLLM_CUTLASS_SRC_DIR})
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(VLLM_CUTLASS_SRC_DIR)
|
||||||
|
if(NOT IS_ABSOLUTE VLLM_CUTLASS_SRC_DIR)
|
||||||
|
get_filename_component(VLLM_CUTLASS_SRC_DIR "${VLLM_CUTLASS_SRC_DIR}" ABSOLUTE)
|
||||||
|
endif()
|
||||||
|
message(STATUS "The VLLM_CUTLASS_SRC_DIR is set, using ${VLLM_CUTLASS_SRC_DIR} for compilation")
|
||||||
|
FetchContent_Declare(cutlass SOURCE_DIR ${VLLM_CUTLASS_SRC_DIR})
|
||||||
|
else()
|
||||||
|
FetchContent_Declare(
|
||||||
|
cutlass
|
||||||
|
GIT_REPOSITORY https://github.com/nvidia/cutlass.git
|
||||||
|
# Please keep this in sync with CUTLASS_REVISION line above.
|
||||||
|
GIT_TAG ${CUTLASS_REVISION}
|
||||||
|
GIT_PROGRESS TRUE
|
||||||
|
|
||||||
|
# Speed up CUTLASS download by retrieving only the specified GIT_TAG instead of the history.
|
||||||
|
# Important: If GIT_SHALLOW is enabled then GIT_TAG works only with branch names and tags.
|
||||||
|
# So if the GIT_TAG above is updated to a commit hash, GIT_SHALLOW must be set to FALSE
|
||||||
|
GIT_SHALLOW TRUE
|
||||||
|
)
|
||||||
|
endif()
|
||||||
|
FetchContent_MakeAvailable(cutlass)
|
||||||
|
|
||||||
list(APPEND VLLM_STABLE_EXT_SRC
|
list(APPEND VLLM_STABLE_EXT_SRC
|
||||||
"csrc/libtorch_stable/cuda_view.cu"
|
|
||||||
"csrc/libtorch_stable/cuda_utils_kernels.cu"
|
|
||||||
"csrc/libtorch_stable/cutlass_extensions/common.cpp"
|
"csrc/libtorch_stable/cutlass_extensions/common.cpp"
|
||||||
"csrc/libtorch_stable/quantization/w8a8/cutlass/scaled_mm_entry.cu"
|
"csrc/libtorch_stable/quantization/w8a8/cutlass/scaled_mm_entry.cu"
|
||||||
"csrc/libtorch_stable/quantization/fp4/nvfp4_quant_entry.cu"
|
"csrc/libtorch_stable/quantization/fp4/nvfp4_quant_entry.cu"
|
||||||
@@ -541,6 +546,14 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
SRCS "${VLLM_STABLE_EXT_SRC}"
|
SRCS "${VLLM_STABLE_EXT_SRC}"
|
||||||
CUDA_ARCHS "${CUDA_ARCHS}")
|
CUDA_ARCHS "${CUDA_ARCHS}")
|
||||||
|
|
||||||
|
if(COOPERATIVE_TOPK_ARCHS)
|
||||||
|
list(APPEND VLLM_STABLE_EXT_SRC
|
||||||
|
"csrc/libtorch_stable/cooperative_topk.cu")
|
||||||
|
set_gencode_flags_for_srcs(
|
||||||
|
SRCS "csrc/libtorch_stable/cooperative_topk.cu"
|
||||||
|
CUDA_ARCHS "${COOPERATIVE_TOPK_ARCHS}")
|
||||||
|
endif()
|
||||||
|
|
||||||
# Only build Marlin kernels if we are building for at least some compatible archs.
|
# Only build Marlin kernels if we are building for at least some compatible archs.
|
||||||
# Keep building Marlin for 9.0 as there are some group sizes and shapes that
|
# Keep building Marlin for 9.0 as there are some group sizes and shapes that
|
||||||
# are not supported by Machete yet.
|
# are not supported by Machete yet.
|
||||||
@@ -886,9 +899,9 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
endif()
|
endif()
|
||||||
|
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 13.0)
|
||||||
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0f" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0f;11.0f" "${CUDA_ARCHS}")
|
||||||
else()
|
else()
|
||||||
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0a;10.3a" "${CUDA_ARCHS}")
|
cuda_archs_loose_intersection(SCALED_MM_ARCHS "10.0a;10.1a;10.3a" "${CUDA_ARCHS}")
|
||||||
endif()
|
endif()
|
||||||
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8 AND SCALED_MM_ARCHS)
|
if(${CMAKE_CUDA_COMPILER_VERSION} VERSION_GREATER_EQUAL 12.8 AND SCALED_MM_ARCHS)
|
||||||
set(CUTLASS_MOE_SM100_SRCS "csrc/libtorch_stable/quantization/w8a8/cutlass/moe/grouped_mm_c3x_sm100.cu")
|
set(CUTLASS_MOE_SM100_SRCS "csrc/libtorch_stable/quantization/w8a8/cutlass/moe/grouped_mm_c3x_sm100.cu")
|
||||||
@@ -958,7 +971,6 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
SRCS "${FP4_SM120_SRCS}"
|
SRCS "${FP4_SM120_SRCS}"
|
||||||
CUDA_ARCHS "${FP4_SM120_ARCHS}")
|
CUDA_ARCHS "${FP4_SM120_ARCHS}")
|
||||||
list(APPEND VLLM_STABLE_EXT_SRC "${FP4_SM120_SRCS}")
|
list(APPEND VLLM_STABLE_EXT_SRC "${FP4_SM120_SRCS}")
|
||||||
target_compile_definitions(_C PRIVATE ENABLE_NVFP4_SM120=1)
|
|
||||||
list(APPEND VLLM_GPU_FLAGS "-DENABLE_NVFP4_SM120=1")
|
list(APPEND VLLM_GPU_FLAGS "-DENABLE_NVFP4_SM120=1")
|
||||||
list(APPEND VLLM_GPU_FLAGS "-DENABLE_CUTLASS_MOE_SM120=1")
|
list(APPEND VLLM_GPU_FLAGS "-DENABLE_CUTLASS_MOE_SM120=1")
|
||||||
message(STATUS "Building SM12x NVFP4 for archs: ${FP4_SM120_ARCHS}")
|
message(STATUS "Building SM12x NVFP4 for archs: ${FP4_SM120_ARCHS}")
|
||||||
@@ -991,7 +1003,6 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
SRCS "${FP4_SM100_SRCS}"
|
SRCS "${FP4_SM100_SRCS}"
|
||||||
CUDA_ARCHS "${FP4_SM100_ARCHS}")
|
CUDA_ARCHS "${FP4_SM100_ARCHS}")
|
||||||
list(APPEND VLLM_STABLE_EXT_SRC "${FP4_SM100_SRCS}")
|
list(APPEND VLLM_STABLE_EXT_SRC "${FP4_SM100_SRCS}")
|
||||||
target_compile_definitions(_C PRIVATE ENABLE_NVFP4_SM100=1)
|
|
||||||
list(APPEND VLLM_GPU_FLAGS "-DENABLE_NVFP4_SM100=1")
|
list(APPEND VLLM_GPU_FLAGS "-DENABLE_NVFP4_SM100=1")
|
||||||
list(APPEND VLLM_GPU_FLAGS "-DENABLE_CUTLASS_MOE_SM100=1")
|
list(APPEND VLLM_GPU_FLAGS "-DENABLE_CUTLASS_MOE_SM100=1")
|
||||||
message(STATUS "Building SM10x/11x NVFP4/MXFP4 for archs: ${FP4_SM100_ARCHS}")
|
message(STATUS "Building SM10x/11x NVFP4/MXFP4 for archs: ${FP4_SM100_ARCHS}")
|
||||||
@@ -1085,25 +1096,24 @@ if(VLLM_GPU_LANG STREQUAL "CUDA" OR VLLM_GPU_LANG STREQUAL "HIP")
|
|||||||
USE_SABI 3
|
USE_SABI 3
|
||||||
WITH_SOABI)
|
WITH_SOABI)
|
||||||
|
|
||||||
|
# Set TORCH_TARGET_VERSION for stable ABI compatibility.
|
||||||
|
# This ensures we only use C-shim APIs available in PyTorch 2.11.
|
||||||
|
# _C_stable_libtorch is abi compatible with PyTorch >= TORCH_TARGET_VERSION
|
||||||
|
# which is currently set to 2.11.
|
||||||
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
|
TORCH_TARGET_VERSION=0x020B000000000000ULL)
|
||||||
|
|
||||||
# Needed to use cuda/hip APIs from C-shim
|
# Needed to use cuda/hip APIs from C-shim
|
||||||
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
||||||
# Set TORCH_TARGET_VERSION for stable ABI compatibility.
|
|
||||||
# This ensures we only use C-shim APIs available in PyTorch 2.11.
|
|
||||||
# _C_stable_libtorch is abi compatible with PyTorch >= TORCH_TARGET_VERSION
|
|
||||||
# which is currently set to 2.11.
|
|
||||||
target_compile_definitions(_C_stable_libtorch PRIVATE
|
|
||||||
TORCH_TARGET_VERSION=0x020B000000000000ULL)
|
|
||||||
target_compile_definitions(_C_stable_libtorch PRIVATE USE_CUDA)
|
target_compile_definitions(_C_stable_libtorch PRIVATE USE_CUDA)
|
||||||
|
if(COOPERATIVE_TOPK_ARCHS)
|
||||||
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
|
VLLM_ENABLE_COOPERATIVE_TOPK=1)
|
||||||
|
endif()
|
||||||
# Needed by CUTLASS kernels
|
# Needed by CUTLASS kernels
|
||||||
target_compile_definitions(_C_stable_libtorch PRIVATE
|
target_compile_definitions(_C_stable_libtorch PRIVATE
|
||||||
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
||||||
elseif(VLLM_GPU_LANG STREQUAL "HIP")
|
elseif(VLLM_GPU_LANG STREQUAL "HIP")
|
||||||
# Set TORCH_TARGET_VERSION for stable ABI compatibility.
|
|
||||||
# This ensures we only use C-shim APIs available in PyTorch 2.10.
|
|
||||||
# _C_stable_libtorch is abi compatible with PyTorch >= TORCH_TARGET_VERSION
|
|
||||||
# which is currently set to 2.10.
|
|
||||||
target_compile_definitions(_C_stable_libtorch PRIVATE
|
|
||||||
TORCH_TARGET_VERSION=0x020A000000000000ULL)
|
|
||||||
target_compile_definitions(_C_stable_libtorch PRIVATE USE_ROCM)
|
target_compile_definitions(_C_stable_libtorch PRIVATE USE_ROCM)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
@@ -1311,25 +1321,20 @@ define_extension_target(
|
|||||||
USE_SABI 3
|
USE_SABI 3
|
||||||
WITH_SOABI)
|
WITH_SOABI)
|
||||||
|
|
||||||
|
# Set TORCH_TARGET_VERSION for stable ABI compatibility.
|
||||||
|
# This ensures we only use C-shim APIs available in PyTorch 2.11.
|
||||||
|
# _moe_C_stable_libtorch is abi compatible with PyTorch >= TORCH_TARGET_VERSION
|
||||||
|
# which is currently set to 2.11.
|
||||||
|
target_compile_definitions(_moe_C_stable_libtorch PRIVATE
|
||||||
|
TORCH_TARGET_VERSION=0x020B000000000000ULL)
|
||||||
|
|
||||||
# Needed to use cuda/hip APIs from C-shim
|
# Needed to use cuda/hip APIs from C-shim
|
||||||
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
if(VLLM_GPU_LANG STREQUAL "CUDA")
|
||||||
# Set TORCH_TARGET_VERSION for stable ABI compatibility.
|
|
||||||
# This ensures we only use C-shim APIs available in PyTorch 2.11.
|
|
||||||
# _moe_C_stable_libtorch is abi compatible with PyTorch >= TORCH_TARGET_VERSION
|
|
||||||
# which is currently set to 2.11.
|
|
||||||
target_compile_definitions(_moe_C_stable_libtorch PRIVATE
|
|
||||||
TORCH_TARGET_VERSION=0x020B000000000000ULL)
|
|
||||||
target_compile_definitions(_moe_C_stable_libtorch PRIVATE USE_CUDA)
|
target_compile_definitions(_moe_C_stable_libtorch PRIVATE USE_CUDA)
|
||||||
# Needed by CUTLASS kernels
|
# Needed by CUTLASS kernels
|
||||||
target_compile_definitions(_moe_C_stable_libtorch PRIVATE
|
target_compile_definitions(_moe_C_stable_libtorch PRIVATE
|
||||||
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
CUTLASS_ENABLE_DIRECT_CUDA_DRIVER_CALL=1)
|
||||||
elseif(VLLM_GPU_LANG STREQUAL "HIP")
|
elseif(VLLM_GPU_LANG STREQUAL "HIP")
|
||||||
# Set TORCH_TARGET_VERSION for stable ABI compatibility.
|
|
||||||
# This ensures we only use C-shim APIs available in PyTorch 2.10.
|
|
||||||
# _moe_C_stable_libtorch is abi compatible with PyTorch >= TORCH_TARGET_VERSION
|
|
||||||
# which is currently set to 2.10.
|
|
||||||
target_compile_definitions(_moe_C_stable_libtorch PRIVATE
|
|
||||||
TORCH_TARGET_VERSION=0x020A000000000000ULL)
|
|
||||||
target_compile_definitions(_moe_C_stable_libtorch PRIVATE USE_ROCM)
|
target_compile_definitions(_moe_C_stable_libtorch PRIVATE USE_ROCM)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
|||||||
@@ -53,6 +53,16 @@ from common import (
|
|||||||
from vllm.v1.worker.workspace import init_workspace_manager
|
from vllm.v1.worker.workspace import init_workspace_manager
|
||||||
|
|
||||||
|
|
||||||
|
def _str2bool(v) -> bool:
|
||||||
|
if isinstance(v, bool):
|
||||||
|
return v
|
||||||
|
if v.lower() in ("true", "1", "yes", "t"):
|
||||||
|
return True
|
||||||
|
if v.lower() in ("false", "0", "no", "f"):
|
||||||
|
return False
|
||||||
|
raise argparse.ArgumentTypeError(f"expected a boolean, got {v!r}")
|
||||||
|
|
||||||
|
|
||||||
def run_standard_attention_benchmark(config: BenchmarkConfig) -> BenchmarkResult:
|
def run_standard_attention_benchmark(config: BenchmarkConfig) -> BenchmarkResult:
|
||||||
"""Run standard attention benchmark (Flash/Triton/FlashInfer)."""
|
"""Run standard attention benchmark (Flash/Triton/FlashInfer)."""
|
||||||
from runner import run_attention_benchmark
|
from runner import run_attention_benchmark
|
||||||
@@ -485,6 +495,20 @@ def main():
|
|||||||
help="Prefill backends to compare (fa2, fa3, fa4). "
|
help="Prefill backends to compare (fa2, fa3, fa4). "
|
||||||
"Uses the first decode backend for impl construction.",
|
"Uses the first decode backend for impl construction.",
|
||||||
)
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--fp8-output-scale",
|
||||||
|
type=float,
|
||||||
|
help="Static per-tensor scale enabling the MLA prefill FP8-output "
|
||||||
|
"comparison on FA4 (fused write vs standalone post-quant).",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--fuse-quant-op",
|
||||||
|
nargs="+",
|
||||||
|
type=_str2bool,
|
||||||
|
help="FP8-output write path(s) to run: false = bf16 attention + "
|
||||||
|
"standalone static-FP8 quant, true = FA4 writes FP8 directly. "
|
||||||
|
"Default: both.",
|
||||||
|
)
|
||||||
|
|
||||||
# Batch specifications
|
# Batch specifications
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
@@ -618,6 +642,12 @@ def main():
|
|||||||
# Prefill backends (e.g., ["fa3", "fa4"])
|
# Prefill backends (e.g., ["fa3", "fa4"])
|
||||||
args.prefill_backends = yaml_config.get("prefill_backends", None)
|
args.prefill_backends = yaml_config.get("prefill_backends", None)
|
||||||
|
|
||||||
|
# FP8 output benchmark knobs; CLI wins.
|
||||||
|
if args.fp8_output_scale is None:
|
||||||
|
args.fp8_output_scale = yaml_config.get("fp8_output_scale", None)
|
||||||
|
if args.fuse_quant_op is None:
|
||||||
|
args.fuse_quant_op = yaml_config.get("fuse_quant_op", None)
|
||||||
|
|
||||||
# Check for special modes
|
# Check for special modes
|
||||||
args.mode = yaml_config.get("mode", None)
|
args.mode = yaml_config.get("mode", None)
|
||||||
|
|
||||||
@@ -787,8 +817,59 @@ def main():
|
|||||||
"skipped (timings are placeholder zeros).[/]"
|
"skipped (timings are placeholder zeros).[/]"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# FA4 fused FP8 output vs standalone post-quant, on the same fa4 kernel:
|
||||||
|
# the delta is the post-quant kernel the fused path removes.
|
||||||
|
fp8_output_scale = getattr(args, "fp8_output_scale", None)
|
||||||
|
if fp8_output_scale is not None:
|
||||||
|
decode_backend = backends[0]
|
||||||
|
fuse_variants = args.fuse_quant_op or [False, True]
|
||||||
|
label_of = {False: "post_quant", True: "fused"}
|
||||||
|
console.print(
|
||||||
|
f"[yellow]FP8 output comparison @ scale={fp8_output_scale} "
|
||||||
|
f"(prefill=fa4, decode impl={decode_backend})[/]"
|
||||||
|
)
|
||||||
|
fp8_results = []
|
||||||
|
total = len(fuse_variants) * len(args.batch_specs)
|
||||||
|
with tqdm(total=total, desc="FP8 output benchmarking") as pbar:
|
||||||
|
for spec in args.batch_specs:
|
||||||
|
for fuse in fuse_variants:
|
||||||
|
config = BenchmarkConfig(
|
||||||
|
backend=decode_backend,
|
||||||
|
batch_spec=spec,
|
||||||
|
num_layers=args.num_layers,
|
||||||
|
head_dim=args.head_dim,
|
||||||
|
num_q_heads=args.num_q_heads,
|
||||||
|
num_kv_heads=args.num_kv_heads,
|
||||||
|
block_size=args.block_size,
|
||||||
|
device=args.device,
|
||||||
|
repeats=args.repeats,
|
||||||
|
warmup_iters=args.warmup_iters,
|
||||||
|
profile_memory=args.profile_memory,
|
||||||
|
kv_cache_dtype=args.kv_cache_dtype,
|
||||||
|
use_cuda_graphs=args.cuda_graphs,
|
||||||
|
prefill_backend="fa4",
|
||||||
|
)
|
||||||
|
result = run_benchmark(
|
||||||
|
config, output_scale=fp8_output_scale, fuse_quant_op=fuse
|
||||||
|
)
|
||||||
|
label = label_of[fuse]
|
||||||
|
labeled_config = replace(result.config, backend=label)
|
||||||
|
result = replace(result, config=labeled_config)
|
||||||
|
fp8_results.append(result)
|
||||||
|
|
||||||
|
if not result.success:
|
||||||
|
console.print(f"[red]Error {label} {spec}: {result.error}[/]")
|
||||||
|
|
||||||
|
pbar.update(1)
|
||||||
|
|
||||||
|
console.print("\n[bold green]FP8 Output Results:[/]")
|
||||||
|
formatter = ResultsFormatter(console)
|
||||||
|
labels = [label_of[f] for f in fuse_variants]
|
||||||
|
formatter.print_table(fp8_results, labels, compare_to_fastest=True)
|
||||||
|
all_results = fp8_results
|
||||||
|
|
||||||
# Handle special mode: decode_vs_prefill comparison
|
# Handle special mode: decode_vs_prefill comparison
|
||||||
if hasattr(args, "mode") and args.mode == "decode_vs_prefill":
|
elif hasattr(args, "mode") and args.mode == "decode_vs_prefill":
|
||||||
console.print("[yellow]Mode: Decode vs Prefill pipeline comparison[/]")
|
console.print("[yellow]Mode: Decode vs Prefill pipeline comparison[/]")
|
||||||
console.print(
|
console.print(
|
||||||
"[dim]For each query length, testing both decode and prefill pipelines[/]"
|
"[dim]For each query length, testing both decode and prefill pipelines[/]"
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# MLA prefill FP8-output microbenchmark (FA4).
|
||||||
|
# Compares the fused FP8 write against bf16 attention + a standalone static-FP8
|
||||||
|
# quant; the delta is the post-quant kernel the fused path removes.
|
||||||
|
# DeepSeek-Coder-V2-Lite dims; FA4 needs SM100/110.
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# python benchmark.py --config configs/mla_fa4_fp8_output.yaml
|
||||||
|
|
||||||
|
description: "MLA prefill FA4 fused-FP8 output vs post-quant"
|
||||||
|
|
||||||
|
model:
|
||||||
|
name: "deepseek-v2-lite"
|
||||||
|
num_layers: 27
|
||||||
|
num_q_heads: 16
|
||||||
|
num_kv_heads: 1
|
||||||
|
head_dim: 576
|
||||||
|
kv_lora_rank: 512
|
||||||
|
qk_nope_head_dim: 128
|
||||||
|
qk_rope_head_dim: 64
|
||||||
|
v_head_dim: 128
|
||||||
|
block_size: 128
|
||||||
|
|
||||||
|
# Pure prefill (q_len == kv_len) so every token goes through forward_mha.
|
||||||
|
batch_specs:
|
||||||
|
- "q512"
|
||||||
|
- "q1k"
|
||||||
|
- "q2k"
|
||||||
|
- "q4k"
|
||||||
|
- "q8k"
|
||||||
|
- "2q4k"
|
||||||
|
- "4q4k"
|
||||||
|
- "8q4k"
|
||||||
|
|
||||||
|
# Only used to construct the MLA impl; the pure-prefill specs skip decode.
|
||||||
|
decode_backends:
|
||||||
|
- CUTLASS_MLA
|
||||||
|
|
||||||
|
# Sweep the two FP8 write paths (prefill backend is fixed to fa4).
|
||||||
|
fp8_output_scale: 0.1
|
||||||
|
fuse_quant_op: [false, true]
|
||||||
|
|
||||||
|
device: "cuda:0"
|
||||||
|
repeats: 50
|
||||||
|
warmup_iters: 10
|
||||||
@@ -708,6 +708,8 @@ def _run_single_benchmark(
|
|||||||
device: torch.device,
|
device: torch.device,
|
||||||
indexer=None,
|
indexer=None,
|
||||||
kv_cache_dtype: str | None = None,
|
kv_cache_dtype: str | None = None,
|
||||||
|
output_scale: float | None = None,
|
||||||
|
fuse_quant_op: bool = False,
|
||||||
) -> BenchmarkResult:
|
) -> BenchmarkResult:
|
||||||
"""
|
"""
|
||||||
Run a single benchmark iteration.
|
Run a single benchmark iteration.
|
||||||
@@ -721,6 +723,11 @@ def _run_single_benchmark(
|
|||||||
mla_dims: MLA dimension configuration
|
mla_dims: MLA dimension configuration
|
||||||
device: Target device
|
device: Target device
|
||||||
indexer: Optional MockIndexer for sparse backends
|
indexer: Optional MockIndexer for sparse backends
|
||||||
|
output_scale: Static per-tensor FP8 scale for prefill output. None
|
||||||
|
keeps the plain bf16 output (no quantization).
|
||||||
|
fuse_quant_op: With output_scale set, True lets the prefill kernel write
|
||||||
|
FP8 directly; False runs bf16 attention then a standalone static-FP8
|
||||||
|
quant. The delta isolates the saved post-quant kernel.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
BenchmarkResult with timing statistics
|
BenchmarkResult with timing statistics
|
||||||
@@ -824,23 +831,55 @@ def _run_single_benchmark(
|
|||||||
num_prefill, mla_dims, query_fmt, device, torch.bfloat16
|
num_prefill, mla_dims, query_fmt, device, torch.bfloat16
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Prefill FP8 output: fused (kernel writes e4m3) vs separate post-quant.
|
||||||
|
prefill_fp8_output = None
|
||||||
|
prefill_output_scale = None
|
||||||
|
prefill_quant_op = None
|
||||||
|
if has_prefill and output_scale is not None:
|
||||||
|
from vllm.platforms import current_platform
|
||||||
|
|
||||||
|
prefill_output_scale = torch.tensor(
|
||||||
|
[output_scale], device=device, dtype=torch.float32
|
||||||
|
)
|
||||||
|
if fuse_quant_op:
|
||||||
|
prefill_fp8_output = torch.empty_like(
|
||||||
|
prefill_inputs["output"], dtype=current_platform.fp8_dtype()
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
from vllm.model_executor.layers.quantization.input_quant_fp8 import (
|
||||||
|
QuantFP8,
|
||||||
|
)
|
||||||
|
from vllm.model_executor.layers.quantization.utils.quant_utils import (
|
||||||
|
GroupShape,
|
||||||
|
)
|
||||||
|
|
||||||
|
prefill_quant_op = QuantFP8(static=True, group_shape=GroupShape.PER_TENSOR)
|
||||||
|
|
||||||
|
fused_output = output_scale is not None and fuse_quant_op
|
||||||
|
|
||||||
# Build forward function (runs a single decode/prefill pass)
|
# Build forward function (runs a single decode/prefill pass)
|
||||||
def forward_fn():
|
def forward_fn():
|
||||||
results = []
|
results = []
|
||||||
if has_decode:
|
if has_decode:
|
||||||
results.append(impl.forward_mqa(decode_inputs, kv_cache, metadata, layer))
|
results.append(impl.forward_mqa(decode_inputs, kv_cache, metadata, layer))
|
||||||
if has_prefill:
|
if has_prefill:
|
||||||
results.append(
|
out = impl.forward_mha(
|
||||||
impl.forward_mha(
|
prefill_inputs["q"],
|
||||||
prefill_inputs["q"],
|
prefill_inputs["k_c_normed"],
|
||||||
prefill_inputs["k_c_normed"],
|
prefill_inputs["k_pe"],
|
||||||
prefill_inputs["k_pe"],
|
kv_cache,
|
||||||
kv_cache,
|
metadata,
|
||||||
metadata,
|
prefill_inputs["k_scale"],
|
||||||
prefill_inputs["k_scale"],
|
prefill_fp8_output if fused_output else prefill_inputs["output"],
|
||||||
prefill_inputs["output"],
|
prefill_output_scale if fused_output else None,
|
||||||
)
|
|
||||||
)
|
)
|
||||||
|
if fused_output:
|
||||||
|
out = prefill_fp8_output
|
||||||
|
elif prefill_quant_op is not None:
|
||||||
|
out, _ = prefill_quant_op(
|
||||||
|
prefill_inputs["output"], prefill_output_scale
|
||||||
|
)
|
||||||
|
results.append(out)
|
||||||
return results[0] if len(results) == 1 else tuple(results)
|
return results[0] if len(results) == 1 else tuple(results)
|
||||||
|
|
||||||
def benchmark_fn():
|
def benchmark_fn():
|
||||||
@@ -881,6 +920,8 @@ def _run_mla_benchmark_batched(
|
|||||||
configs_with_params: list[tuple], # [(config, threshold, num_splits), ...]
|
configs_with_params: list[tuple], # [(config, threshold, num_splits), ...]
|
||||||
index_topk: int = 2048,
|
index_topk: int = 2048,
|
||||||
prefill_backend: str | None = None,
|
prefill_backend: str | None = None,
|
||||||
|
output_scale: float | None = None,
|
||||||
|
fuse_quant_op: bool = False,
|
||||||
) -> list[BenchmarkResult]:
|
) -> list[BenchmarkResult]:
|
||||||
"""
|
"""
|
||||||
Unified batched MLA benchmark runner for all backends.
|
Unified batched MLA benchmark runner for all backends.
|
||||||
@@ -1020,6 +1061,8 @@ def _run_mla_benchmark_batched(
|
|||||||
device,
|
device,
|
||||||
indexer=indexer,
|
indexer=indexer,
|
||||||
kv_cache_dtype=kv_cache_dtype,
|
kv_cache_dtype=kv_cache_dtype,
|
||||||
|
output_scale=output_scale,
|
||||||
|
fuse_quant_op=fuse_quant_op,
|
||||||
)
|
)
|
||||||
results.append(result)
|
results.append(result)
|
||||||
|
|
||||||
@@ -1047,6 +1090,8 @@ def run_mla_benchmark(
|
|||||||
num_kv_splits: int | None = None,
|
num_kv_splits: int | None = None,
|
||||||
index_topk: int = 2048,
|
index_topk: int = 2048,
|
||||||
prefill_backend: str | None = None,
|
prefill_backend: str | None = None,
|
||||||
|
output_scale: float | None = None,
|
||||||
|
fuse_quant_op: bool = False,
|
||||||
) -> BenchmarkResult | list[BenchmarkResult]:
|
) -> BenchmarkResult | list[BenchmarkResult]:
|
||||||
"""
|
"""
|
||||||
Unified MLA benchmark runner for all backends.
|
Unified MLA benchmark runner for all backends.
|
||||||
@@ -1066,6 +1111,9 @@ def run_mla_benchmark(
|
|||||||
index_topk: Topk value for sparse MLA backends (default 2048)
|
index_topk: Topk value for sparse MLA backends (default 2048)
|
||||||
prefill_backend: Prefill backend name (e.g., "fa3", "fa4").
|
prefill_backend: Prefill backend name (e.g., "fa3", "fa4").
|
||||||
When set, forces the specified FlashAttention version for prefill.
|
When set, forces the specified FlashAttention version for prefill.
|
||||||
|
output_scale: Static per-tensor FP8 scale for prefill output (None = bf16).
|
||||||
|
fuse_quant_op: With output_scale set, fuse the FP8 write into the prefill
|
||||||
|
kernel vs a standalone post-quant kernel. See _run_single_benchmark.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
BenchmarkResult (single mode) or list of BenchmarkResult (batched mode)
|
BenchmarkResult (single mode) or list of BenchmarkResult (batched mode)
|
||||||
@@ -1090,7 +1138,12 @@ def run_mla_benchmark(
|
|||||||
|
|
||||||
# Use unified batched execution
|
# Use unified batched execution
|
||||||
results = _run_mla_benchmark_batched(
|
results = _run_mla_benchmark_batched(
|
||||||
backend, configs_with_params, index_topk, prefill_backend=prefill_backend
|
backend,
|
||||||
|
configs_with_params,
|
||||||
|
index_topk,
|
||||||
|
prefill_backend=prefill_backend,
|
||||||
|
output_scale=output_scale,
|
||||||
|
fuse_quant_op=fuse_quant_op,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Return single result or list based on input
|
# Return single result or list based on input
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ from dataclasses import dataclass, field
|
|||||||
import aiohttp
|
import aiohttp
|
||||||
import huggingface_hub.constants
|
import huggingface_hub.constants
|
||||||
from tqdm.asyncio import tqdm
|
from tqdm.asyncio import tqdm
|
||||||
from transformers import AutoTokenizer, PreTrainedTokenizer, PreTrainedTokenizerFast
|
from transformers import AutoTokenizer, PythonBackend, TokenizersBackend
|
||||||
|
|
||||||
# NOTE(simon): do not import vLLM here so the benchmark script
|
# NOTE(simon): do not import vLLM here so the benchmark script
|
||||||
# can run without vLLM installed.
|
# can run without vLLM installed.
|
||||||
@@ -609,7 +609,7 @@ def get_tokenizer(
|
|||||||
tokenizer_mode: str = "auto",
|
tokenizer_mode: str = "auto",
|
||||||
trust_remote_code: bool = False,
|
trust_remote_code: bool = False,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
) -> PreTrainedTokenizer | PreTrainedTokenizerFast:
|
) -> PythonBackend | TokenizersBackend:
|
||||||
if pretrained_model_name_or_path is not None and not os.path.exists(
|
if pretrained_model_name_or_path is not None and not os.path.exists(
|
||||||
pretrained_model_name_or_path
|
pretrained_model_name_or_path
|
||||||
):
|
):
|
||||||
|
|||||||
@@ -0,0 +1,358 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
"""Benchmark and regression-test pinned (page-locked) CPU memory for vLLM.
|
||||||
|
|
||||||
|
Verifies that enabling pinned memory does not regress throughput or latency
|
||||||
|
compared to unpinned memory. Each condition runs in an isolated ``spawn``
|
||||||
|
subprocess so both start from a cold CUDA context, giving an unbiased
|
||||||
|
comparison.
|
||||||
|
|
||||||
|
Usage
|
||||||
|
-----
|
||||||
|
Run all tests with the default model::
|
||||||
|
|
||||||
|
python benchmarks/benchmark_pin_memory.py -v
|
||||||
|
|
||||||
|
Override the model and optional max-model-len::
|
||||||
|
|
||||||
|
python benchmarks/benchmark_pin_memory.py --model unsloth/Qwen3-1.7B -v
|
||||||
|
python benchmarks/benchmark_pin_memory.py --model unsloth/Qwen3-1.7B \
|
||||||
|
--max-model-len 8192 -v
|
||||||
|
|
||||||
|
Run only throughput or latency tests::
|
||||||
|
|
||||||
|
python benchmarks/benchmark_pin_memory.py -v -k test_throughput
|
||||||
|
python benchmarks/benchmark_pin_memory.py -v -k test_latency
|
||||||
|
|
||||||
|
Run only the v1 or v2 runner variant::
|
||||||
|
|
||||||
|
python benchmarks/benchmark_pin_memory.py -v -k v1
|
||||||
|
python benchmarks/benchmark_pin_memory.py -v -k v2
|
||||||
|
|
||||||
|
Note: on WSL2, v1 runner tests are skipped because pin memory is not available
|
||||||
|
for the v1 runner without cpu_offload_gb. Run on other platforms to exercise v1.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import multiprocessing
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
# Allow up to 2% degradation. Both benchmark runs start from an identical
|
||||||
|
# cold CUDA context (separate spawn subprocesses), so the measured difference
|
||||||
|
# reflects the genuine pin_memory overhead rather than cold/warm ordering bias.
|
||||||
|
_THROUGHPUT_TOLERANCE = 0.98
|
||||||
|
_THROUGHPUT_NUM_REQUESTS = 200
|
||||||
|
_THROUGHPUT_INPUT_LEN = 128
|
||||||
|
_THROUGHPUT_OUTPUT_LEN = 512
|
||||||
|
_THROUGHPUT_MAX_NUM_SEQS = 128
|
||||||
|
|
||||||
|
# Latency benchmark constants — match latency.py defaults.
|
||||||
|
_LATENCY_TOLERANCE = 1.02 # Allow up to 2% latency regression.
|
||||||
|
_LATENCY_BATCH_SIZE = 64
|
||||||
|
_LATENCY_INPUT_LEN = 32
|
||||||
|
_LATENCY_OUTPUT_LEN = 128
|
||||||
|
_LATENCY_WARMUP_ITERS = 5
|
||||||
|
_LATENCY_BENCH_ITERS = 15
|
||||||
|
|
||||||
|
_DEFAULT_MODEL = "unsloth/Qwen3-1.7B"
|
||||||
|
_DEFAULT_MAX_MODEL_LEN = 16384
|
||||||
|
|
||||||
|
|
||||||
|
def _benchmark_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(add_help=False)
|
||||||
|
parser.add_argument("--model", default=_DEFAULT_MODEL)
|
||||||
|
parser.add_argument("--max-model-len", type=int, default=_DEFAULT_MAX_MODEL_LEN)
|
||||||
|
args, _ = parser.parse_known_args()
|
||||||
|
return args
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def model() -> str:
|
||||||
|
return _benchmark_args().model
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def max_model_len() -> int:
|
||||||
|
return _benchmark_args().max_model_len
|
||||||
|
|
||||||
|
|
||||||
|
def _skip_if_pin_memory_not_available(engine_args_kwargs: dict) -> None:
|
||||||
|
"""Skip the current pytest test if pin_memory is unavailable for this config."""
|
||||||
|
import vllm.utils.platform_utils as pu
|
||||||
|
from vllm.config import set_current_vllm_config
|
||||||
|
from vllm.engine.arg_utils import EngineArgs
|
||||||
|
|
||||||
|
vllm_config = EngineArgs(**engine_args_kwargs).create_engine_config()
|
||||||
|
with set_current_vllm_config(vllm_config):
|
||||||
|
pu.is_pin_memory_available.cache_clear()
|
||||||
|
if not pu.is_pin_memory_available():
|
||||||
|
import os
|
||||||
|
|
||||||
|
runner = "v2" if os.environ.get("VLLM_USE_V2_MODEL_RUNNER") == "1" else "v1"
|
||||||
|
model = engine_args_kwargs.get("model", "unknown")
|
||||||
|
print(
|
||||||
|
f"\033[33mSKIP: pin_memory not available for "
|
||||||
|
f"{runner} runner, model={model}\033[0m"
|
||||||
|
)
|
||||||
|
pytest.skip("pin_memory not available for this configuration")
|
||||||
|
|
||||||
|
|
||||||
|
def _throughput_worker(
|
||||||
|
pin: bool,
|
||||||
|
engine_args_kwargs: dict,
|
||||||
|
q: "multiprocessing.Queue[float]",
|
||||||
|
v2_mode: bool = False,
|
||||||
|
) -> None:
|
||||||
|
"""Run throughput benchmark in a fresh spawn subprocess.
|
||||||
|
|
||||||
|
Delegates to vllm/benchmarks/throughput.py main() using the random dataset,
|
||||||
|
so the methodology matches the official benchmark. Results are written to a
|
||||||
|
temp JSON file and forwarded through the queue as tokens/s.
|
||||||
|
|
||||||
|
v2_mode: when True, monkeypatches is_uva_available() to always return True
|
||||||
|
so the v2 model runner's UVA buffers remain functional even when pin=False.
|
||||||
|
This isolates the non-UVA pin_memory paths in v2.
|
||||||
|
"""
|
||||||
|
import vllm.utils.platform_utils as pu
|
||||||
|
from vllm.platforms import current_platform
|
||||||
|
|
||||||
|
pu.is_pin_memory_available.cache_clear()
|
||||||
|
pu.is_uva_available.cache_clear()
|
||||||
|
type(current_platform).is_pin_memory_available = classmethod(lambda cls: pin)
|
||||||
|
if v2_mode:
|
||||||
|
pu.is_uva_available = lambda: True
|
||||||
|
|
||||||
|
from vllm.benchmarks.throughput import add_cli_args
|
||||||
|
from vllm.benchmarks.throughput import main as throughput_main
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
add_cli_args(parser)
|
||||||
|
args = parser.parse_args([])
|
||||||
|
|
||||||
|
for key, val in engine_args_kwargs.items():
|
||||||
|
setattr(args, key, val)
|
||||||
|
args.max_num_seqs = _THROUGHPUT_MAX_NUM_SEQS
|
||||||
|
args.dataset_name = "random"
|
||||||
|
args.input_len = _THROUGHPUT_INPUT_LEN
|
||||||
|
args.output_len = _THROUGHPUT_OUTPUT_LEN
|
||||||
|
# Nullify defaults that conflict with explicit input/output_len.
|
||||||
|
args.random_input_len = None
|
||||||
|
args.random_output_len = None
|
||||||
|
args.random_prefix_len = None
|
||||||
|
args.num_prompts = _THROUGHPUT_NUM_REQUESTS
|
||||||
|
args.seed = 0
|
||||||
|
args.disable_detokenize = True
|
||||||
|
|
||||||
|
with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f:
|
||||||
|
tmp_path = f.name
|
||||||
|
args.output_json = tmp_path
|
||||||
|
|
||||||
|
throughput_main(args)
|
||||||
|
|
||||||
|
with open(tmp_path) as f:
|
||||||
|
results = json.load(f)
|
||||||
|
q.put(results["tokens_per_second"])
|
||||||
|
|
||||||
|
|
||||||
|
def _run_throughput_benchmark(
|
||||||
|
pin: bool,
|
||||||
|
engine_args_kwargs: dict,
|
||||||
|
v2_mode: bool = False,
|
||||||
|
) -> float:
|
||||||
|
ctx = multiprocessing.get_context("spawn")
|
||||||
|
q = ctx.Queue()
|
||||||
|
p = ctx.Process(
|
||||||
|
target=_throughput_worker,
|
||||||
|
args=(pin, engine_args_kwargs, q, v2_mode),
|
||||||
|
)
|
||||||
|
p.start()
|
||||||
|
p.join()
|
||||||
|
if p.exitcode != 0:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"Throughput benchmark subprocess (pin={pin}) exited with code {p.exitcode}"
|
||||||
|
)
|
||||||
|
return q.get()
|
||||||
|
|
||||||
|
|
||||||
|
def _latency_worker(
|
||||||
|
pin: bool,
|
||||||
|
engine_args_kwargs: dict,
|
||||||
|
q: "multiprocessing.Queue[dict]",
|
||||||
|
v2_mode: bool = False,
|
||||||
|
) -> None:
|
||||||
|
"""Run latency benchmark in a fresh spawn subprocess.
|
||||||
|
|
||||||
|
Follows latency.py methodology: fixed batch of dummy token IDs, warmup
|
||||||
|
iterations to reach steady state, then timed iterations reduced to avg
|
||||||
|
and percentiles. Results are written to a temp JSON file by latency_main
|
||||||
|
and forwarded through the queue.
|
||||||
|
"""
|
||||||
|
import vllm.utils.platform_utils as pu
|
||||||
|
from vllm.platforms import current_platform
|
||||||
|
|
||||||
|
pu.is_pin_memory_available.cache_clear()
|
||||||
|
pu.is_uva_available.cache_clear()
|
||||||
|
type(current_platform).is_pin_memory_available = classmethod(lambda cls: pin)
|
||||||
|
if v2_mode:
|
||||||
|
pu.is_uva_available = lambda: True
|
||||||
|
|
||||||
|
from vllm.benchmarks.latency import add_cli_args
|
||||||
|
from vllm.benchmarks.latency import main as latency_main
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
add_cli_args(parser)
|
||||||
|
args = parser.parse_args([])
|
||||||
|
|
||||||
|
for key, val in engine_args_kwargs.items():
|
||||||
|
setattr(args, key, val)
|
||||||
|
args.input_len = _LATENCY_INPUT_LEN
|
||||||
|
args.output_len = _LATENCY_OUTPUT_LEN
|
||||||
|
args.batch_size = _LATENCY_BATCH_SIZE
|
||||||
|
args.num_iters_warmup = _LATENCY_WARMUP_ITERS
|
||||||
|
args.num_iters = _LATENCY_BENCH_ITERS
|
||||||
|
args.profile = False
|
||||||
|
args.disable_detokenize = True
|
||||||
|
|
||||||
|
with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f:
|
||||||
|
tmp_path = f.name
|
||||||
|
args.output_json = tmp_path
|
||||||
|
|
||||||
|
latency_main(args)
|
||||||
|
|
||||||
|
with open(tmp_path) as f:
|
||||||
|
results = json.load(f)
|
||||||
|
q.put(results)
|
||||||
|
|
||||||
|
|
||||||
|
def _run_latency_benchmark(
|
||||||
|
pin: bool,
|
||||||
|
engine_args_kwargs: dict,
|
||||||
|
v2_mode: bool = False,
|
||||||
|
) -> dict:
|
||||||
|
ctx = multiprocessing.get_context("spawn")
|
||||||
|
q = ctx.Queue()
|
||||||
|
p = ctx.Process(
|
||||||
|
target=_latency_worker,
|
||||||
|
args=(pin, engine_args_kwargs, q, v2_mode),
|
||||||
|
)
|
||||||
|
p.start()
|
||||||
|
p.join()
|
||||||
|
if p.exitcode != 0:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"Latency benchmark subprocess (pin={pin}) exited with code {p.exitcode}"
|
||||||
|
)
|
||||||
|
return q.get()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"test_v2_runner",
|
||||||
|
[
|
||||||
|
pytest.param(False, id="v1"),
|
||||||
|
pytest.param(True, id="v2"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
class TestPinnedMemory:
|
||||||
|
"""Verify pinned memory yields >= throughput vs unpinned via real vLLM inference."""
|
||||||
|
|
||||||
|
def test_throughput(self, monkeypatch, test_v2_runner, model, max_model_len):
|
||||||
|
"""Benchmark throughput with pin_memory forced on then off.
|
||||||
|
|
||||||
|
Delegates to vllm/benchmarks/throughput.py main() with the random
|
||||||
|
dataset. Each condition runs in an isolated spawn subprocess so both
|
||||||
|
start from a cold CUDA context, giving an unbiased comparison.
|
||||||
|
"""
|
||||||
|
monkeypatch.setenv("VLLM_ENABLE_V1_MULTIPROCESSING", "0")
|
||||||
|
monkeypatch.setenv("VLLM_USE_V2_MODEL_RUNNER", "1" if test_v2_runner else "0")
|
||||||
|
|
||||||
|
engine_args_kwargs = dict(
|
||||||
|
model=model,
|
||||||
|
gpu_memory_utilization=0.88,
|
||||||
|
max_model_len=max_model_len,
|
||||||
|
enable_prefix_caching=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
_skip_if_pin_memory_not_available(engine_args_kwargs)
|
||||||
|
|
||||||
|
unpinned_tps = _run_throughput_benchmark(
|
||||||
|
False, engine_args_kwargs, v2_mode=test_v2_runner
|
||||||
|
)
|
||||||
|
pinned_tps = _run_throughput_benchmark(
|
||||||
|
True, engine_args_kwargs, v2_mode=test_v2_runner
|
||||||
|
)
|
||||||
|
|
||||||
|
pct_diff = (pinned_tps - unpinned_tps) / unpinned_tps * 100
|
||||||
|
runner = "v2" if test_v2_runner else "v1"
|
||||||
|
print(
|
||||||
|
f"\n=== Throughput results ({runner} runner, {model}) ==="
|
||||||
|
f"\npin_memory=True: {pinned_tps:.1f} tok/s"
|
||||||
|
f"\npin_memory=False: {unpinned_tps:.1f} tok/s"
|
||||||
|
f"\nDifference: {pct_diff:+.1f}% (pinned vs unpinned)"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert pinned_tps >= unpinned_tps * _THROUGHPUT_TOLERANCE, (
|
||||||
|
f"Pinned throughput ({pinned_tps:.1f} tok/s) fell more than "
|
||||||
|
f"{(1.0 - _THROUGHPUT_TOLERANCE) * 100:.1f}% below "
|
||||||
|
f"unpinned ({unpinned_tps:.1f} tok/s)."
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_latency(self, monkeypatch, test_v2_runner, model, max_model_len):
|
||||||
|
"""Benchmark per-batch latency with pin_memory forced on then off.
|
||||||
|
|
||||||
|
Follows vllm/benchmarks/latency.py: fixed dummy-token batch, warmup
|
||||||
|
iterations to reach steady state, then timed iterations reduced to avg
|
||||||
|
and percentiles. Subprocesses run serially so each gets a cold CUDA
|
||||||
|
context without GPU memory pressure from the other run.
|
||||||
|
"""
|
||||||
|
monkeypatch.setenv("VLLM_ENABLE_V1_MULTIPROCESSING", "0")
|
||||||
|
monkeypatch.setenv("VLLM_USE_V2_MODEL_RUNNER", "1" if test_v2_runner else "0")
|
||||||
|
|
||||||
|
engine_args_kwargs = dict(
|
||||||
|
model=model,
|
||||||
|
gpu_memory_utilization=0.88,
|
||||||
|
max_model_len=max_model_len,
|
||||||
|
enable_prefix_caching=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
_skip_if_pin_memory_not_available(engine_args_kwargs)
|
||||||
|
|
||||||
|
unpinned = _run_latency_benchmark(
|
||||||
|
False, engine_args_kwargs, v2_mode=test_v2_runner
|
||||||
|
)
|
||||||
|
pinned = _run_latency_benchmark(
|
||||||
|
True, engine_args_kwargs, v2_mode=test_v2_runner
|
||||||
|
)
|
||||||
|
|
||||||
|
pct_diff = (
|
||||||
|
(pinned["avg_latency"] - unpinned["avg_latency"])
|
||||||
|
/ unpinned["avg_latency"]
|
||||||
|
* 100
|
||||||
|
)
|
||||||
|
runner = "v2" if test_v2_runner else "v1"
|
||||||
|
print(
|
||||||
|
f"\n=== Latency results ({runner} runner, {model}) ==="
|
||||||
|
f"\npin_memory=True: avg={pinned['avg_latency']:.3f}s"
|
||||||
|
f" p50={pinned['percentiles']['50']:.3f}s"
|
||||||
|
f" p99={pinned['percentiles']['99']:.3f}s"
|
||||||
|
f"\npin_memory=False: avg={unpinned['avg_latency']:.3f}s"
|
||||||
|
f" p50={unpinned['percentiles']['50']:.3f}s"
|
||||||
|
f" p99={unpinned['percentiles']['99']:.3f}s"
|
||||||
|
f"\nDifference: {pct_diff:+.1f}% (pinned vs unpinned)"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert pinned["avg_latency"] <= unpinned["avg_latency"] * _LATENCY_TOLERANCE, (
|
||||||
|
f"Pinned avg latency ({pinned['avg_latency']:.3f}s) exceeded "
|
||||||
|
f"unpinned ({unpinned['avg_latency']:.3f}s) by more than "
|
||||||
|
f"{(_LATENCY_TOLERANCE - 1.0) * 100:.1f}%."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
_parser = argparse.ArgumentParser(add_help=False)
|
||||||
|
_parser.add_argument("--model", default=_DEFAULT_MODEL)
|
||||||
|
_parser.add_argument("--max-model-len", type=int, default=_DEFAULT_MAX_MODEL_LEN)
|
||||||
|
_, _remaining = _parser.parse_known_args()
|
||||||
|
sys.exit(pytest.main([__file__] + _remaining))
|
||||||
@@ -17,7 +17,7 @@ from vllm.model_executor.layers.fused_moe.fused_flydsl_moe import fused_flydsl_m
|
|||||||
from vllm.model_executor.layers.quantization.compressed_tensors.compressed_tensors_moe import ( # noqa: E501
|
from vllm.model_executor.layers.quantization.compressed_tensors.compressed_tensors_moe import ( # noqa: E501
|
||||||
compressed_tensors_moe_w4a16_flydsl,
|
compressed_tensors_moe_w4a16_flydsl,
|
||||||
)
|
)
|
||||||
from vllm.platforms import current_platform
|
from vllm.utils.platform_utils import get_device_name_as_file_name
|
||||||
|
|
||||||
RoutingBuffers = tuple[
|
RoutingBuffers = tuple[
|
||||||
torch.Tensor, # sorted_token_ids
|
torch.Tensor, # sorted_token_ids
|
||||||
@@ -259,7 +259,7 @@ def tune_flydsl_moe_w4a16(
|
|||||||
)
|
)
|
||||||
us_best = us
|
us_best = us
|
||||||
tuned_config[str(num_tokens)] = tile_config
|
tuned_config[str(num_tokens)] = tile_config
|
||||||
device_name = current_platform.get_device_name().replace(" ", "_")
|
device_name = get_device_name_as_file_name()
|
||||||
tuned_config_file_name = (
|
tuned_config_file_name = (
|
||||||
f"E={num_experts},N={inter_dim},device_name={device_name},"
|
f"E={num_experts},N={inter_dim},device_name={device_name},"
|
||||||
f"dtype=int4_w4a16,backend=flydsl.json"
|
f"dtype=int4_w4a16,backend=flydsl.json"
|
||||||
|
|||||||
@@ -80,13 +80,17 @@ _FI_MAX_SIZES = {
|
|||||||
2: 64 * MiB, # 64MB
|
2: 64 * MiB, # 64MB
|
||||||
4: 64 * MiB, # 64MB
|
4: 64 * MiB, # 64MB
|
||||||
8: 64 * MiB, # 64MB
|
8: 64 * MiB, # 64MB
|
||||||
|
16: 64 * MiB, # 64MB (multi-node)
|
||||||
}
|
}
|
||||||
|
|
||||||
# Global workspace tensors for FlashInfer (keyed by backend name)
|
# Global workspace tensors for FlashInfer (keyed by backend name)
|
||||||
_FI_WORKSPACES: dict = {}
|
_FI_WORKSPACES: dict = {}
|
||||||
|
|
||||||
# Backends to benchmark
|
# Backends to benchmark. trtllm is single-node only and can hang cross-node, so
|
||||||
FLASHINFER_BACKENDS = ["trtllm", "mnnvl"]
|
# multi-node sweeps can restrict to mnnvl via FI_BACKENDS=mnnvl.
|
||||||
|
FLASHINFER_BACKENDS = [
|
||||||
|
b for b in os.environ.get("FI_BACKENDS", "trtllm,mnnvl").split(",") if b
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def setup_flashinfer_workspace(
|
def setup_flashinfer_workspace(
|
||||||
@@ -995,7 +999,10 @@ def main():
|
|||||||
rank = int(os.environ["RANK"])
|
rank = int(os.environ["RANK"])
|
||||||
world_size = int(os.environ["WORLD_SIZE"])
|
world_size = int(os.environ["WORLD_SIZE"])
|
||||||
|
|
||||||
device = torch.device(f"cuda:{rank}")
|
# Use LOCAL_RANK for the device so multi-node runs (global rank >= GPUs per
|
||||||
|
# node) map to a valid local GPU; falls back to global rank single-node.
|
||||||
|
local_rank = int(os.environ.get("LOCAL_RANK", rank))
|
||||||
|
device = torch.device(f"cuda:{local_rank}")
|
||||||
torch.accelerator.set_device_index(device)
|
torch.accelerator.set_device_index(device)
|
||||||
torch.set_default_device(device)
|
torch.set_default_device(device)
|
||||||
|
|
||||||
|
|||||||
@@ -391,16 +391,19 @@ def get_configs_compute_bound(use_fp16, block_quant_shape) -> list[dict[str, int
|
|||||||
config = dict(zip(keys, config_values))
|
config = dict(zip(keys, config_values))
|
||||||
configs.append(config)
|
configs.append(config)
|
||||||
|
|
||||||
# Remove configs that are not compatible with fp8 block quantization
|
# Drop configs incompatible with fp8 block quantization. A tile must align
|
||||||
# BLOCK_SIZE_K must be a multiple of block_k
|
# to the quant-block scale grid, i.e. tile and block must divide one
|
||||||
# BLOCK_SIZE_N must be a multiple of block_n
|
# another. The kernel indexes scales per element (offs_bn // group_n,
|
||||||
|
# k_start // group_k), so a tile narrower than the block (e.g. N=64 with
|
||||||
|
# block_n=128) is valid -- and often faster at small batch. An exact
|
||||||
|
# multiple was required before, which dropped those smaller tiles entirely.
|
||||||
if block_quant_shape is not None and not use_fp16:
|
if block_quant_shape is not None and not use_fp16:
|
||||||
block_n, block_k = block_quant_shape[0], block_quant_shape[1]
|
block_n, block_k = block_quant_shape[0], block_quant_shape[1]
|
||||||
for config in configs[:]:
|
for config in configs[:]:
|
||||||
if (
|
bn, bk = config["BLOCK_SIZE_N"], config["BLOCK_SIZE_K"]
|
||||||
config["BLOCK_SIZE_K"] % block_k != 0
|
n_aligned = bn % block_n == 0 or block_n % bn == 0
|
||||||
or config["BLOCK_SIZE_N"] % block_n != 0
|
k_aligned = bk % block_k == 0 or block_k % bk == 0
|
||||||
):
|
if not (n_aligned and k_aligned):
|
||||||
configs.remove(config)
|
configs.remove(config)
|
||||||
return configs
|
return configs
|
||||||
|
|
||||||
|
|||||||
@@ -19,13 +19,11 @@ from vllm.utils.torch_utils import (
|
|||||||
logger = init_logger(__name__)
|
logger = init_logger(__name__)
|
||||||
|
|
||||||
NUM_BLOCKS = 128 * 1024
|
NUM_BLOCKS = 128 * 1024
|
||||||
PARTITION_SIZE = 512
|
|
||||||
PARTITION_SIZE_ROCM = 256
|
PARTITION_SIZE_ROCM = 256
|
||||||
|
|
||||||
|
|
||||||
@torch.inference_mode()
|
@torch.inference_mode()
|
||||||
def main(
|
def main(
|
||||||
version: str,
|
|
||||||
num_seqs: int,
|
num_seqs: int,
|
||||||
seq_len: int,
|
seq_len: int,
|
||||||
num_query_heads: int,
|
num_query_heads: int,
|
||||||
@@ -82,27 +80,20 @@ def main(
|
|||||||
|
|
||||||
# Prepare for the paged attention kernel.
|
# Prepare for the paged attention kernel.
|
||||||
output = torch.empty_like(query)
|
output = torch.empty_like(query)
|
||||||
if version == "v2":
|
num_partitions = (max_seq_len + PARTITION_SIZE_ROCM - 1) // PARTITION_SIZE_ROCM
|
||||||
if current_platform.is_rocm():
|
tmp_output = torch.empty(
|
||||||
global PARTITION_SIZE
|
size=(num_seqs, num_query_heads, num_partitions, head_size),
|
||||||
if not args.custom_paged_attn and not current_platform.is_navi():
|
dtype=output.dtype,
|
||||||
PARTITION_SIZE = 1024
|
device=output.device,
|
||||||
else:
|
)
|
||||||
PARTITION_SIZE = PARTITION_SIZE_ROCM
|
exp_sums = torch.empty(
|
||||||
num_partitions = (max_seq_len + PARTITION_SIZE - 1) // PARTITION_SIZE
|
size=(num_seqs, num_query_heads, num_partitions),
|
||||||
tmp_output = torch.empty(
|
dtype=torch.float32,
|
||||||
size=(num_seqs, num_query_heads, num_partitions, head_size),
|
device=output.device,
|
||||||
dtype=output.dtype,
|
)
|
||||||
device=output.device,
|
max_logits = torch.empty_like(exp_sums)
|
||||||
)
|
|
||||||
exp_sums = torch.empty(
|
|
||||||
size=(num_seqs, num_query_heads, num_partitions),
|
|
||||||
dtype=torch.float32,
|
|
||||||
device=output.device,
|
|
||||||
)
|
|
||||||
max_logits = torch.empty_like(exp_sums)
|
|
||||||
|
|
||||||
def run_cuda_benchmark(num_iters: int, profile: bool = False) -> float:
|
def run_benchmark(num_iters: int, profile: bool = False) -> float:
|
||||||
torch.accelerator.synchronize()
|
torch.accelerator.synchronize()
|
||||||
if profile:
|
if profile:
|
||||||
torch.cuda.cudart().cudaProfilerStart()
|
torch.cuda.cudart().cudaProfilerStart()
|
||||||
@@ -112,67 +103,26 @@ def main(
|
|||||||
k_scale = v_scale = torch.tensor(1.0, dtype=torch.float32, device=device)
|
k_scale = v_scale = torch.tensor(1.0, dtype=torch.float32, device=device)
|
||||||
|
|
||||||
for _ in range(num_iters):
|
for _ in range(num_iters):
|
||||||
if version == "v1":
|
ops.paged_attention_rocm(
|
||||||
ops.paged_attention_v1(
|
output,
|
||||||
output,
|
exp_sums,
|
||||||
query,
|
max_logits,
|
||||||
key_cache,
|
tmp_output,
|
||||||
value_cache,
|
query,
|
||||||
num_kv_heads,
|
key_cache,
|
||||||
scale,
|
value_cache,
|
||||||
block_tables,
|
num_kv_heads,
|
||||||
seq_lens,
|
scale,
|
||||||
block_size,
|
block_tables,
|
||||||
max_seq_len,
|
seq_lens,
|
||||||
alibi_slopes,
|
None,
|
||||||
kv_cache_dtype,
|
block_size,
|
||||||
k_scale,
|
max_seq_len,
|
||||||
v_scale,
|
alibi_slopes,
|
||||||
)
|
kv_cache_dtype,
|
||||||
elif version == "v2":
|
k_scale,
|
||||||
if not args.custom_paged_attn:
|
v_scale,
|
||||||
ops.paged_attention_v2(
|
)
|
||||||
output,
|
|
||||||
exp_sums,
|
|
||||||
max_logits,
|
|
||||||
tmp_output,
|
|
||||||
query,
|
|
||||||
key_cache,
|
|
||||||
value_cache,
|
|
||||||
num_kv_heads,
|
|
||||||
scale,
|
|
||||||
block_tables,
|
|
||||||
seq_lens,
|
|
||||||
block_size,
|
|
||||||
max_seq_len,
|
|
||||||
alibi_slopes,
|
|
||||||
kv_cache_dtype,
|
|
||||||
k_scale,
|
|
||||||
v_scale,
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
ops.paged_attention_rocm(
|
|
||||||
output,
|
|
||||||
exp_sums,
|
|
||||||
max_logits,
|
|
||||||
tmp_output,
|
|
||||||
query,
|
|
||||||
key_cache,
|
|
||||||
value_cache,
|
|
||||||
num_kv_heads,
|
|
||||||
scale,
|
|
||||||
block_tables,
|
|
||||||
seq_lens,
|
|
||||||
None,
|
|
||||||
block_size,
|
|
||||||
max_seq_len,
|
|
||||||
alibi_slopes,
|
|
||||||
kv_cache_dtype,
|
|
||||||
k_scale,
|
|
||||||
v_scale,
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
raise ValueError(f"Invalid version: {version}")
|
|
||||||
torch.accelerator.synchronize()
|
torch.accelerator.synchronize()
|
||||||
|
|
||||||
end_time = time.perf_counter()
|
end_time = time.perf_counter()
|
||||||
@@ -182,7 +132,6 @@ def main(
|
|||||||
|
|
||||||
# Warmup.
|
# Warmup.
|
||||||
print("Warming up...")
|
print("Warming up...")
|
||||||
run_benchmark = run_cuda_benchmark
|
|
||||||
run_benchmark(num_iters=3, profile=False)
|
run_benchmark(num_iters=3, profile=False)
|
||||||
|
|
||||||
# Benchmark.
|
# Benchmark.
|
||||||
@@ -195,12 +144,13 @@ def main(
|
|||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"This script benchmarks the paged attention kernel. "
|
"This script benchmarks the ROCm paged attention kernel. "
|
||||||
"By default this is no longer used in vLLM inference."
|
"By default this is no longer used in vLLM inference."
|
||||||
)
|
)
|
||||||
|
if not current_platform.is_rocm():
|
||||||
|
raise RuntimeError("This benchmark requires the ROCm platform.")
|
||||||
|
|
||||||
parser = FlexibleArgumentParser(description="Benchmark the paged attention kernel.")
|
parser = FlexibleArgumentParser(description="Benchmark the paged attention kernel.")
|
||||||
parser.add_argument("--version", type=str, choices=["v1", "v2"], default="v2")
|
|
||||||
parser.add_argument("--batch-size", type=int, default=8)
|
parser.add_argument("--batch-size", type=int, default=8)
|
||||||
parser.add_argument("--seq-len", type=int, default=4096)
|
parser.add_argument("--seq-len", type=int, default=4096)
|
||||||
parser.add_argument("--num-query-heads", type=int, default=64)
|
parser.add_argument("--num-query-heads", type=int, default=64)
|
||||||
@@ -208,7 +158,7 @@ if __name__ == "__main__":
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--head-size",
|
"--head-size",
|
||||||
type=int,
|
type=int,
|
||||||
choices=[64, 80, 96, 112, 120, 128, 192, 256],
|
choices=[64, 128],
|
||||||
default=128,
|
default=128,
|
||||||
)
|
)
|
||||||
parser.add_argument("--block-size", type=int, choices=[16, 32], default=16)
|
parser.add_argument("--block-size", type=int, choices=[16, 32], default=16)
|
||||||
@@ -224,11 +174,7 @@ if __name__ == "__main__":
|
|||||||
choices=["auto", "fp8", "fp8_e5m2", "fp8_e4m3"],
|
choices=["auto", "fp8", "fp8_e5m2", "fp8_e4m3"],
|
||||||
default="auto",
|
default="auto",
|
||||||
help="Data type for kv cache storage. If 'auto', will use model "
|
help="Data type for kv cache storage. If 'auto', will use model "
|
||||||
"data type. CUDA 11.8+ supports fp8 (=fp8_e4m3) and fp8_e5m2. "
|
"data type. ROCm (AMD GPU) supports fp8 (=fp8_e4m3)",
|
||||||
"ROCm (AMD GPU) supports fp8 (=fp8_e4m3)",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--custom-paged-attn", action="store_true", help="Use custom paged attention"
|
|
||||||
)
|
)
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
print(args)
|
print(args)
|
||||||
@@ -236,7 +182,6 @@ if __name__ == "__main__":
|
|||||||
if args.num_query_heads % args.num_kv_heads != 0:
|
if args.num_query_heads % args.num_kv_heads != 0:
|
||||||
raise ValueError("num_query_heads must be divisible by num_kv_heads")
|
raise ValueError("num_query_heads must be divisible by num_kv_heads")
|
||||||
main(
|
main(
|
||||||
version=args.version,
|
|
||||||
num_seqs=args.batch_size,
|
num_seqs=args.batch_size,
|
||||||
seq_len=args.seq_len,
|
seq_len=args.seq_len,
|
||||||
num_query_heads=args.num_query_heads,
|
num_query_heads=args.num_query_heads,
|
||||||
|
|||||||
@@ -0,0 +1,108 @@
|
|||||||
|
# SPDX-License-Identifier: Apache-2.0
|
||||||
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||||
|
|
||||||
|
# Benchmark ReLUSquaredActivation: custom CUDA kernel vs forward_native, both
|
||||||
|
# eager and under torch.compile (Inductor fuses relu+square into one kernel).
|
||||||
|
|
||||||
|
import itertools
|
||||||
|
|
||||||
|
import torch
|
||||||
|
import torch.nn.functional as F
|
||||||
|
|
||||||
|
import vllm.model_executor.layers.activation # noqa: F401
|
||||||
|
from vllm.benchmarks.lib.utils import default_vllm_config
|
||||||
|
from vllm.triton_utils import triton
|
||||||
|
from vllm.utils.argparse_utils import FlexibleArgumentParser
|
||||||
|
from vllm.utils.torch_utils import STR_DTYPE_TO_TORCH_DTYPE, set_random_seed
|
||||||
|
|
||||||
|
# Capped so the largest tensor stays under 2**31 elements: the shared activation
|
||||||
|
# kernel computes the per-token pointer offset (blockIdx.x * d) in 32-bit, which
|
||||||
|
# overflows for tensors with >2**32 elements. Realistic token counts are well
|
||||||
|
# below this; the kernel-vs-native gap is already clear at these sizes.
|
||||||
|
batch_size_range = [1, 16, 128]
|
||||||
|
seq_len_range = [1, 16, 64, 1024]
|
||||||
|
intermediate_size = [3072, 9728, 12288]
|
||||||
|
configs = list(itertools.product(batch_size_range, seq_len_range, intermediate_size))
|
||||||
|
|
||||||
|
|
||||||
|
@default_vllm_config()
|
||||||
|
def benchmark_relu_squared(
|
||||||
|
batch_size: int,
|
||||||
|
seq_len: int,
|
||||||
|
intermediate_size: int,
|
||||||
|
provider: str,
|
||||||
|
dtype: torch.dtype,
|
||||||
|
):
|
||||||
|
device = "cuda"
|
||||||
|
num_tokens = batch_size * seq_len
|
||||||
|
set_random_seed(42)
|
||||||
|
torch.set_default_device(device)
|
||||||
|
|
||||||
|
x = torch.randn(num_tokens, intermediate_size, dtype=dtype, device=device)
|
||||||
|
out = torch.empty_like(x)
|
||||||
|
|
||||||
|
def native(x: torch.Tensor) -> torch.Tensor:
|
||||||
|
return torch.square(F.relu(x))
|
||||||
|
|
||||||
|
# Verify the custom kernel matches the native implementation before timing.
|
||||||
|
ref = native(x)
|
||||||
|
torch.ops._C.relu_squared(out, x)
|
||||||
|
torch.testing.assert_close(out, ref)
|
||||||
|
|
||||||
|
if provider == "custom":
|
||||||
|
# Custom CUDA kernel — single fused kernel.
|
||||||
|
fn = lambda: torch.ops._C.relu_squared(out, x)
|
||||||
|
elif provider == "native":
|
||||||
|
# forward_native, eager — relu and square as separate ops.
|
||||||
|
fn = lambda: native(x)
|
||||||
|
elif provider == "native_compiled":
|
||||||
|
# forward_native under torch.compile — Inductor fuses relu+square.
|
||||||
|
# This is the real production baseline (custom ops are off when
|
||||||
|
# Inductor is enabled), so it is the comparison reviewers care about.
|
||||||
|
compiled = torch.compile(native)
|
||||||
|
compiled(x) # warm up / trigger compilation before timing
|
||||||
|
fn = lambda: compiled(x)
|
||||||
|
|
||||||
|
ms, min_ms, max_ms = triton.testing.do_bench_cudagraph(
|
||||||
|
fn, quantiles=[0.5, 0.2, 0.8]
|
||||||
|
)
|
||||||
|
return ms, max_ms, min_ms
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = FlexibleArgumentParser(
|
||||||
|
description="Benchmark ReLUSquaredActivation: custom kernel vs native."
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--dtype",
|
||||||
|
type=str,
|
||||||
|
choices=["half", "bfloat16", "float"],
|
||||||
|
default="bfloat16",
|
||||||
|
)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
dtype = STR_DTYPE_TO_TORCH_DTYPE[args.dtype]
|
||||||
|
|
||||||
|
perf_report = triton.testing.perf_report(
|
||||||
|
triton.testing.Benchmark(
|
||||||
|
x_names=["batch_size", "seq_len", "intermediate_size"],
|
||||||
|
x_vals=configs,
|
||||||
|
line_arg="provider",
|
||||||
|
line_vals=["custom", "native_compiled", "native"],
|
||||||
|
line_names=[
|
||||||
|
"Custom Kernel",
|
||||||
|
"Native (torch.compile)",
|
||||||
|
"Native (eager)",
|
||||||
|
],
|
||||||
|
styles=[("blue", "-"), ("green", "-"), ("red", "-")],
|
||||||
|
ylabel="ms",
|
||||||
|
plot_name="relu_squared-eager-performance",
|
||||||
|
args={},
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
perf_report(
|
||||||
|
lambda batch_size, seq_len, intermediate_size, provider: benchmark_relu_squared(
|
||||||
|
batch_size, seq_len, intermediate_size, provider, dtype
|
||||||
|
)
|
||||||
|
).run(print_data=True)
|
||||||
@@ -19,6 +19,7 @@ from vllm.model_executor.layers.quantization.utils.fp8_utils import (
|
|||||||
from vllm.platforms import current_platform
|
from vllm.platforms import current_platform
|
||||||
from vllm.triton_utils import triton
|
from vllm.triton_utils import triton
|
||||||
from vllm.utils.argparse_utils import FlexibleArgumentParser
|
from vllm.utils.argparse_utils import FlexibleArgumentParser
|
||||||
|
from vllm.utils.platform_utils import get_device_name_as_file_name
|
||||||
|
|
||||||
mp.set_start_method("spawn", force=True)
|
mp.set_start_method("spawn", force=True)
|
||||||
|
|
||||||
@@ -264,7 +265,7 @@ def save_configs(
|
|||||||
input_type="fp8",
|
input_type="fp8",
|
||||||
) -> None:
|
) -> None:
|
||||||
os.makedirs(save_path, exist_ok=True)
|
os.makedirs(save_path, exist_ok=True)
|
||||||
device_name = current_platform.get_device_name().replace(" ", "_")
|
device_name = get_device_name_as_file_name()
|
||||||
json_file_name = (
|
json_file_name = (
|
||||||
f"N={N},K={K},device_name={device_name},dtype={input_type}_w8a8,"
|
f"N={N},K={K},device_name={device_name},dtype={input_type}_w8a8,"
|
||||||
f"block_shape=[{block_n},{block_k}].json"
|
f"block_shape=[{block_n},{block_k}].json"
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import time
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
import torch
|
import torch
|
||||||
|
|
||||||
|
from vllm.platforms import CpuArchEnum, current_platform
|
||||||
from vllm.utils.argparse_utils import FlexibleArgumentParser
|
from vllm.utils.argparse_utils import FlexibleArgumentParser
|
||||||
from vllm.utils.torch_utils import set_random_seed
|
from vllm.utils.torch_utils import set_random_seed
|
||||||
|
|
||||||
@@ -14,17 +15,15 @@ from vllm.utils.torch_utils import set_random_seed
|
|||||||
try:
|
try:
|
||||||
from vllm._custom_ops import cpu_fused_moe, cpu_prepack_moe_weight
|
from vllm._custom_ops import cpu_fused_moe, cpu_prepack_moe_weight
|
||||||
except (ImportError, AttributeError) as e:
|
except (ImportError, AttributeError) as e:
|
||||||
print("ERROR: CPU fused MoE operations are not available on this platform.")
|
|
||||||
print("This benchmark requires x86 CPU with proper vLLM CPU extensions compiled.")
|
|
||||||
print(
|
|
||||||
"The cpu_fused_moe kernel is typically available on Linux x86_64 "
|
|
||||||
"with AVX2/AVX512."
|
|
||||||
)
|
|
||||||
print(f"Import error: {e}")
|
print(f"Import error: {e}")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
# ISA selection following test_cpu_fused_moe.py pattern
|
# ISA selection following test_cpu_fused_moe.py pattern
|
||||||
ISA_CHOICES = ["amx", "vec"] if torch.cpu._is_amx_tile_supported() else ["vec"]
|
ISA_CHOICES = ["vec"]
|
||||||
|
if torch.cpu._is_amx_tile_supported():
|
||||||
|
ISA_CHOICES.append("amx")
|
||||||
|
if current_platform.get_cpu_architecture() == CpuArchEnum.ARM:
|
||||||
|
ISA_CHOICES.append("neon")
|
||||||
|
|
||||||
|
|
||||||
@torch.inference_mode()
|
@torch.inference_mode()
|
||||||
@@ -145,7 +144,7 @@ if __name__ == "__main__":
|
|||||||
"--isa",
|
"--isa",
|
||||||
type=str,
|
type=str,
|
||||||
choices=ISA_CHOICES,
|
choices=ISA_CHOICES,
|
||||||
default=ISA_CHOICES[0],
|
default="vec",
|
||||||
help=f"ISA to use (available: {ISA_CHOICES})",
|
help=f"ISA to use (available: {ISA_CHOICES})",
|
||||||
)
|
)
|
||||||
parser.add_argument("--seed", type=int, default=0)
|
parser.add_argument("--seed", type=int, default=0)
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user