forked from tinygrad/tinygrad
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
147fd0e2c6 | ||
|
|
1ecb99480e | ||
|
|
77b5e6774e | ||
|
|
f1041dc0ac | ||
|
|
47e0c43976 | ||
|
|
0f776c6e46 | ||
|
|
e0139fafc1 | ||
|
|
218225e8d0 | ||
|
|
9096d7cc2e | ||
|
|
066d25f5fb | ||
|
|
cd6aeebfee | ||
|
|
e537e895b1 | ||
|
|
9ab06dffad | ||
|
|
12435a2dab | ||
|
|
8f5f57c7d9 | ||
|
|
1ecf403294 | ||
|
|
fd51ecf983 | ||
|
|
b5afa3848e | ||
|
|
822eab057f | ||
|
|
7ac74d1550 | ||
|
|
772a8dfe31 | ||
|
|
08e62454b6 | ||
|
|
a2ae56674a | ||
|
|
dccdd190aa | ||
|
|
9205527db0 | ||
|
|
cab034b863 | ||
|
|
4300ebc455 | ||
|
|
7596c1b8f5 | ||
|
|
001b3710d3 | ||
|
|
a62dc9ceb5 | ||
|
|
464c56862f | ||
|
|
ac96d98745 | ||
|
|
89be3590aa | ||
|
|
95ad047445 | ||
|
|
e625c27598 | ||
|
|
6ec96f6088 | ||
|
|
9471157346 | ||
|
|
36c753bd63 | ||
|
|
b27470b6db | ||
|
|
03ef5197fc | ||
|
|
965bd194f2 | ||
|
|
af90dc00de | ||
|
|
f12e2a75db | ||
|
|
caae46cfba | ||
|
|
1309cea247 | ||
|
|
cbdc13279d | ||
|
|
c8dfd10257 | ||
|
|
88ce63a49a | ||
|
|
5977df267f | ||
|
|
f2c3a72b0c | ||
|
|
9b66c2b0b7 | ||
|
|
658b96cbfb | ||
|
|
b86ad6053a | ||
|
|
502e613c9c | ||
|
|
840d2bf1ea | ||
|
|
8a1c3dc1bf | ||
|
|
e0694fdb8e | ||
|
|
678f83e41b | ||
|
|
a11b686c71 | ||
|
|
a0cbbc35ad | ||
|
|
fe94453d52 | ||
|
|
f793cdeb87 | ||
|
|
1bcea19846 | ||
|
|
c1cc277fc3 | ||
|
|
2551a60d97 | ||
|
|
e7aa26ed29 | ||
|
|
cf8232ec6a | ||
|
|
658c566e22 | ||
|
|
a8a9ac0e95 | ||
|
|
250f05a776 | ||
|
|
da9425c1a7 | ||
|
|
ae51bdd06a | ||
|
|
80d99d52a5 | ||
|
|
375ee2c576 | ||
|
|
1dc500426e | ||
|
|
585bd95b50 | ||
|
|
6af29b913b | ||
|
|
baab7e334d | ||
|
|
51420d1f99 | ||
|
|
43bce1f39f | ||
|
|
9f9a8b0b5b | ||
|
|
6e6059dde0 | ||
|
|
20d98b19c3 | ||
|
|
bb5671a837 | ||
|
|
be05028419 | ||
|
|
615ec6acf0 | ||
|
|
c4732a18bd | ||
|
|
5986d656a2 | ||
|
|
fc2bd53700 | ||
|
|
89ec2b3a74 | ||
|
|
84fc34b274 | ||
|
|
28edea5d67 | ||
|
|
2653147cb7 | ||
|
|
0774575442 | ||
|
|
a65ec5c693 | ||
|
|
b6835f4134 | ||
|
|
3b0b3a2e64 | ||
|
|
9448924d9e | ||
|
|
c5a1f9f5f9 | ||
|
|
ee0382ad99 | ||
|
|
d5058427ea | ||
|
|
6f26603f06 | ||
|
|
7e0b14243e | ||
|
|
942022c309 | ||
|
|
e701106a64 | ||
|
|
291a19650b | ||
|
|
ad49f8148b | ||
|
|
da1f46ff3f | ||
|
|
1e567a5cf8 | ||
|
|
9e7103647d | ||
|
|
4a756a37d8 | ||
|
|
60b6dca5ba | ||
|
|
84597ed53c | ||
|
|
2e19354c1c | ||
|
|
d06226b575 | ||
|
|
a7cb80bfab | ||
|
|
a6d59a0b45 | ||
|
|
eb3bc277b3 | ||
|
|
239f9a3029 | ||
|
|
b465c17b56 | ||
|
|
945cc46475 | ||
|
|
648e5bb223 | ||
|
|
a2345787b9 | ||
|
|
12c4963489 | ||
|
|
403fdfcfd4 | ||
|
|
22674798df | ||
|
|
75ce11593c | ||
|
|
fe774a4319 | ||
|
|
8ad5f9e74f | ||
|
|
ea7672931f | ||
|
|
514d2a0774 | ||
|
|
7b48f3cc45 | ||
|
|
a5484b767e | ||
|
|
b4509fba31 | ||
|
|
0f25b4b289 | ||
|
|
f664bcc8bd | ||
|
|
1af05dae77 | ||
|
|
76e8a3250c | ||
|
|
0c015a24fe | ||
|
|
a1881b0c17 | ||
|
|
1b1978b9c0 | ||
|
|
c1e85f699c | ||
|
|
1823a5043f | ||
|
|
46e8ea15c1 | ||
|
|
1216fff781 | ||
|
|
6ad9a688ed | ||
|
|
74b04f7dca | ||
|
|
69857d0ab0 | ||
|
|
a976ace404 | ||
|
|
4b60121498 | ||
|
|
b5f31d7505 | ||
|
|
865d5796f8 | ||
|
|
e74be4a140 | ||
|
|
394dc24110 | ||
|
|
9f2b69b870 | ||
|
|
0b534f71c2 | ||
|
|
b087663c35 | ||
|
|
940a8d5ba9 | ||
|
|
d290e77a5b | ||
|
|
23d310bcc1 | ||
|
|
1e8945a28c | ||
|
|
c7849ac593 | ||
|
|
0f82d92b9d | ||
|
|
4c63f7e786 | ||
|
|
0047bcc535 | ||
|
|
f203d8b221 | ||
|
|
a6dd5a224b | ||
|
|
bf99de7b1e | ||
|
|
9cd365c12e | ||
|
|
16a65b4fd0 | ||
|
|
2d24af888b | ||
|
|
1b58ef0d60 | ||
|
|
17d36d0952 | ||
|
|
7b3912d8e4 | ||
|
|
98163832e4 | ||
|
|
37beef6de3 | ||
|
|
f21851b099 | ||
|
|
ec177c80c2 | ||
|
|
13a25b2e67 | ||
|
|
5d9035f5a6 | ||
|
|
0f804c9a83 | ||
|
|
0eee93f0c0 | ||
|
|
583553f467 | ||
|
|
6fc6b51b59 | ||
|
|
d1c868f990 | ||
|
|
2fcd55583f | ||
|
|
8b48e19ce2 | ||
|
|
3770dd9d80 | ||
|
|
5b649616ff | ||
|
|
9a64fc0d28 | ||
|
|
3e0e0290ce | ||
|
|
2f8ac77c25 | ||
|
|
89bed28716 | ||
|
|
74ee305948 | ||
|
|
f198a9e1ba | ||
|
|
ac3d457d5e | ||
|
|
60e52fbe36 | ||
|
|
6c95b1f39d | ||
|
|
6ba8bf282f | ||
|
|
689ab9151b | ||
|
|
adc8c3b28f | ||
|
|
154d114364 | ||
|
|
fe96c8d345 | ||
|
|
f205352cd7 | ||
|
|
90b1c0dd96 | ||
|
|
42748ccb92 | ||
|
|
05e91a248d | ||
|
|
714500edfd | ||
|
|
57ad46c6e4 | ||
|
|
e02da8f5ac | ||
|
|
0662946fac | ||
|
|
da52006bde | ||
|
|
1c1b4d14e9 | ||
|
|
4204edc60b | ||
|
|
8def8145e4 | ||
|
|
4c9a930de2 | ||
|
|
26247573e1 | ||
|
|
f2eb92948d | ||
|
|
a128fa0f8a | ||
|
|
969a1b35ca | ||
|
|
9ef319f349 | ||
|
|
080b26e7d7 | ||
|
|
44558a37f7 | ||
|
|
2c397eb2a2 | ||
|
|
a83f219253 | ||
|
|
a95159d579 | ||
|
|
9cf5e66899 | ||
|
|
b4a4817c9c | ||
|
|
de1d562b69 | ||
|
|
c9ef5d8fe5 | ||
|
|
e8c595c29e | ||
|
|
360980f1a3 | ||
|
|
109c63b904 | ||
|
|
7129419500 | ||
|
|
4ff7f20b9d | ||
|
|
86c5c969ea | ||
|
|
6a56d3c859 | ||
|
|
ab6b0d3a21 | ||
|
|
2a7310ab59 | ||
|
|
73b25bf47d | ||
|
|
2a0caa09c2 | ||
|
|
881709cd33 | ||
|
|
39aae679e4 | ||
|
|
af935e7d32 | ||
|
|
f522e83a02 | ||
|
|
d95d018bb5 | ||
|
|
05275c9ec3 | ||
|
|
8e508a9927 | ||
|
|
3a480b858f | ||
|
|
32d69d07d7 | ||
|
|
d55d829635 | ||
|
|
c38f6ce140 | ||
|
|
c2689c505e | ||
|
|
cdfa0f29fd | ||
|
|
baf3b60cfb | ||
|
|
9513f025c5 | ||
|
|
b899392f30 | ||
|
|
7ae6898e31 | ||
|
|
3291e00df7 | ||
|
|
9d2f2b8e34 | ||
|
|
9915bcf2b4 | ||
|
|
76c87d81b3 | ||
|
|
fd2e4f2353 | ||
|
|
29469577e8 | ||
|
|
a982480512 | ||
|
|
e01a3eb59a | ||
|
|
cf925d1ac5 | ||
|
|
b252f890da | ||
|
|
292cb6ae26 | ||
|
|
250cb10e8f | ||
|
|
ed90de6583 | ||
|
|
29f0886395 | ||
|
|
b98f1881ef | ||
|
|
6f1cf717de | ||
|
|
0104b16b9b | ||
|
|
f5eb46a3d9 | ||
|
|
8b2e0930d7 | ||
|
|
74411984fc | ||
|
|
d2cd269e28 | ||
|
|
17cec8d645 | ||
|
|
476a2a0a96 | ||
|
|
38ecefaacb | ||
|
|
0e778296be | ||
|
|
6c9d8c7e41 | ||
|
|
1400ce105f | ||
|
|
154c865966 | ||
|
|
e8945c74de | ||
|
|
45c7252aed | ||
|
|
6146c64d81 | ||
|
|
ad7c8c21ea | ||
|
|
02a7b7fe48 | ||
|
|
2f145a98e0 | ||
|
|
5f4eeb054c | ||
|
|
680ce54dd4 | ||
|
|
fffce0a6b4 | ||
|
|
51b88b2265 | ||
|
|
b54cb272d0 | ||
|
|
d21e34e617 | ||
|
|
5a4b244e6b | ||
|
|
a6fd96f620 | ||
|
|
b03ceb806e | ||
|
|
25e0b725d1 | ||
|
|
1aba668a37 | ||
|
|
b53a266254 | ||
|
|
461e9becec | ||
|
|
9569fdfa36 | ||
|
|
8365c28cd5 | ||
|
|
4762a24022 | ||
|
|
57c7e0a8f8 | ||
|
|
393c6b236c | ||
|
|
4756971c88 | ||
|
|
5e794be8af | ||
|
|
73c8dae60d | ||
|
|
dc4dd898b7 | ||
|
|
bb1f376ae6 | ||
|
|
7e06d3ebba | ||
|
|
bb59eed82f | ||
|
|
cc038b31b6 | ||
|
|
a531a649fb | ||
|
|
8d703a6369 | ||
|
|
0dad6cc518 | ||
|
|
cff1065f5e | ||
|
|
ef05178855 | ||
|
|
87707ef0b8 | ||
|
|
825f148469 | ||
|
|
f82b16a0e9 | ||
|
|
7487c13b61 | ||
|
|
54c15d74a4 | ||
|
|
dbbc261075 | ||
|
|
f1108f1cbe | ||
|
|
812f485cd7 | ||
|
|
3c5b8bf50c | ||
|
|
525f80e0d2 | ||
|
|
edffc246ed | ||
|
|
7733c217c5 | ||
|
|
d917895569 | ||
|
|
158506b91e | ||
|
|
328bfe6b9b | ||
|
|
5b12764b83 | ||
|
|
53655a4ee5 | ||
|
|
6b808c5fe6 | ||
|
|
2a72b00679 | ||
|
|
c7b03457d7 | ||
|
|
494bb12500 | ||
|
|
419e997187 | ||
|
|
84d2d047ea | ||
|
|
122a50fe8c | ||
|
|
e555748807 | ||
|
|
f732f66709 | ||
|
|
82e037aad5 | ||
|
|
146c31586d | ||
|
|
df1c183e46 | ||
|
|
d01e3d7719 | ||
|
|
b63bd02969 | ||
|
|
57e8bf61e8 | ||
|
|
72e010d816 | ||
|
|
f1bd06134d | ||
|
|
ef0ef705fe | ||
|
|
d8855ec266 | ||
|
|
b8a74c1569 | ||
|
|
a388d2cb1a | ||
|
|
65397bfdeb | ||
|
|
ae0edc8a67 | ||
|
|
e1fef895b1 | ||
|
|
3a9db08b49 | ||
|
|
bdb3afd566 | ||
|
|
9fcc87761e | ||
|
|
1353250b6c | ||
|
|
60d7db093e | ||
|
|
525c20dc7e | ||
|
|
75ff9b7a9a | ||
|
|
15b166ce6d | ||
|
|
943236ef74 | ||
|
|
25b1bc8eff | ||
|
|
34a05b31fe | ||
|
|
d09c0f28c5 | ||
|
|
12a910f1d2 | ||
|
|
98ecab7563 | ||
|
|
02054b53fe | ||
|
|
1591e4f66b | ||
|
|
d1ae30f7ef | ||
|
|
d5bc27797b | ||
|
|
4b7904eca9 | ||
|
|
bcafa72b7f | ||
|
|
d2316ba91a | ||
|
|
b1d1816f43 | ||
|
|
19d9d29b7e | ||
|
|
6410dcb7c2 | ||
|
|
92df52d79a | ||
|
|
0c392089d9 | ||
|
|
fbca6183ad | ||
|
|
b2a95d32bb | ||
|
|
0695e322a8 | ||
|
|
e3a3764917 | ||
|
|
51ed6e94b2 | ||
|
|
0757a9a819 | ||
|
|
2fc0bd150b | ||
|
|
aac3dceaf6 | ||
|
|
a12d0933c1 | ||
|
|
25091951ba | ||
|
|
62376c8b2b | ||
|
|
647965fb09 | ||
|
|
0fad07c684 | ||
|
|
81e33b8439 | ||
|
|
68b0ad05a4 | ||
|
|
e80c8a7548 | ||
|
|
b5a3b8de20 | ||
|
|
a2f502b89e | ||
|
|
0766616962 | ||
|
|
1f3950a484 | ||
|
|
544eb2c402 | ||
|
|
9ad6a56d17 | ||
|
|
e5ef9ec5b1 | ||
|
|
3a83b56da5 | ||
|
|
520e2e0727 | ||
|
|
acb700fc26 | ||
|
|
20cd7177de | ||
|
|
b07f962058 | ||
|
|
66593f135f | ||
|
|
e76211fcbc | ||
|
|
400ad93892 | ||
|
|
3ef0e5e01e | ||
|
|
52ebed991e | ||
|
|
d4eba5800d | ||
|
|
78610b681e | ||
|
|
3989f5b559 | ||
|
|
73d479a016 | ||
|
|
e306650d39 | ||
|
|
d8a7a1c9c7 | ||
|
|
3730172c10 | ||
|
|
0e266f376c | ||
|
|
0599e86186 | ||
|
|
5a84d86db7 | ||
|
|
fb96394ff5 | ||
|
|
bb67829e99 | ||
|
|
84b249ef0e | ||
|
|
5d66a2d885 | ||
|
|
9789337722 | ||
|
|
21e6926a6a | ||
|
|
551560b87c | ||
|
|
0e420e68b4 | ||
|
|
ef53a6fc19 | ||
|
|
499f50483b | ||
|
|
5b73076e48 | ||
|
|
58d13a6e3e | ||
|
|
71fcb23d4a | ||
|
|
82e955fe79 | ||
|
|
14faf7a5c0 | ||
|
|
5e76eff26d | ||
|
|
5fde033794 | ||
|
|
1c6c42715f | ||
|
|
50cc7175cb | ||
|
|
239091d111 | ||
|
|
2bd1fff79c | ||
|
|
1781d5bced | ||
|
|
9182948951 | ||
|
|
11213398b9 | ||
|
|
ebbcdd6577 | ||
|
|
73ca0e870c | ||
|
|
d40f5b766b | ||
|
|
75b58fe2d3 | ||
|
|
56861852be | ||
|
|
ef71acc88a | ||
|
|
35ddfc3d39 | ||
|
|
97187bf8b6 | ||
|
|
f326df8ae8 | ||
|
|
c66935f7b9 | ||
|
|
801be5f7b9 | ||
|
|
10ac427aaa | ||
|
|
2b1844da27 | ||
|
|
f37b836618 | ||
|
|
1630c87d0e | ||
|
|
48ec5efad9 | ||
|
|
581b2388c2 | ||
|
|
c6c16b2946 | ||
|
|
8658a97197 | ||
|
|
6ef3270fc8 | ||
|
|
66c5206b42 | ||
|
|
478e758755 | ||
|
|
51b7c40788 | ||
|
|
0123c394e5 | ||
|
|
8423c06144 | ||
|
|
38dcadf07b | ||
|
|
ee4f696086 | ||
|
|
12c7b1bb01 | ||
|
|
8435d2d23b | ||
|
|
e00858a2c3 | ||
|
|
433581f8ed | ||
|
|
3b41a04b96 | ||
|
|
290521f68e | ||
|
|
870f63d9cc | ||
|
|
4c2d4f683a | ||
|
|
a340723bf1 | ||
|
|
ce7163e9b4 | ||
|
|
f08299d2ec | ||
|
|
5dcc4c7f1b | ||
|
|
f8e2dd4dd1 | ||
|
|
e0da644171 | ||
|
|
9b6f1b86cb | ||
|
|
3e1c04bcdf | ||
|
|
ab413ce72f | ||
|
|
f461ccf407 | ||
|
|
4fcea8493d | ||
|
|
2b5a73ac65 | ||
|
|
7f3df6ea21 | ||
|
|
f5404ca53c | ||
|
|
677220ae7e | ||
|
|
431666da74 | ||
|
|
30eb42a69e | ||
|
|
da61b40604 | ||
|
|
be364a1adb | ||
|
|
52166fd7eb | ||
|
|
dc8501af30 | ||
|
|
8c720e8760 | ||
|
|
70ce29b630 | ||
|
|
560df206cc | ||
|
|
4996bb668b | ||
|
|
9dee724fc4 | ||
|
|
09106e4aae | ||
|
|
fb71d1e5fd | ||
|
|
ca7574cb2d | ||
|
|
e213b85810 | ||
|
|
35f37a64a9 | ||
|
|
572a3c15c6 | ||
|
|
5cf42dc4db | ||
|
|
b13e071463 | ||
|
|
edc8b99853 | ||
|
|
ed2f45712b | ||
|
|
a5f2b4872a | ||
|
|
63e930fec3 | ||
|
|
d0e739453e | ||
|
|
55e4bdd353 | ||
|
|
1877eddde4 | ||
|
|
5ed262982a | ||
|
|
6d53cac457 | ||
|
|
68e83b850f | ||
|
|
86e908db57 | ||
|
|
033184b3cb | ||
|
|
53eff8970a | ||
|
|
d1d0960e6e | ||
|
|
8a2846b31a | ||
|
|
d16cc6c012 | ||
|
|
1b73993521 | ||
|
|
e921fb44ee | ||
|
|
69dd1817d0 | ||
|
|
f750c15965 | ||
|
|
550cf2ca7f | ||
|
|
b977ec0813 | ||
|
|
897254ad6c | ||
|
|
74040663bf | ||
|
|
0dfca4e74b | ||
|
|
7c21271a5f | ||
|
|
6a40216724 | ||
|
|
965ea59b16 | ||
|
|
a9f07c31bc | ||
|
|
0a53e72f70 | ||
|
|
27c9ed5a84 | ||
|
|
c7bb561ef9 | ||
|
|
d9560a631c | ||
|
|
a19d689481 | ||
|
|
f32f3464d6 | ||
|
|
1c6e43c203 | ||
|
|
7e68045fb2 | ||
|
|
020abe0556 | ||
|
|
2004c9757d | ||
|
|
c1eeb3b99c | ||
|
|
75d380a77c | ||
|
|
61e4dc6ad5 | ||
|
|
d3252ccd85 | ||
|
|
0bacd9fc9b | ||
|
|
af89be317e | ||
|
|
632c2fb119 | ||
|
|
c27b99d68f | ||
|
|
9aff00a6ea | ||
|
|
c86ee5bfaf | ||
|
|
a4f05ebd1a | ||
|
|
bf0d055b39 | ||
|
|
0bc34c000f | ||
|
|
561318fea7 | ||
|
|
0838021753 | ||
|
|
cf9d8c8142 | ||
|
|
c6e342cdac | ||
|
|
26d03a86a1 | ||
|
|
b2cc06218a | ||
|
|
afad7d0cd1 | ||
|
|
30e72d5820 | ||
|
|
d8e1e4dc61 | ||
|
|
75678b2cbe | ||
|
|
394c2d1db1 | ||
|
|
fa695ac1ce | ||
|
|
b9b438c516 | ||
|
|
bb55a3001f | ||
|
|
e8289c75b1 | ||
|
|
134cf56904 | ||
|
|
ea1be2e4cd | ||
|
|
53853ae49b | ||
|
|
874c1db4af | ||
|
|
17ecaf4682 | ||
|
|
54be477152 | ||
|
|
5f8fe9a331 | ||
|
|
4e8370309c | ||
|
|
6d6f0dada7 | ||
|
|
60dd9a162c | ||
|
|
beb5982165 | ||
|
|
cb5295168d | ||
|
|
fd579433bc | ||
|
|
44816218b5 | ||
|
|
4006366752 | ||
|
|
7f90497efc | ||
|
|
e4afdf9ea1 | ||
|
|
e9789d8a70 | ||
|
|
884eb53e89 | ||
|
|
d39365809a | ||
|
|
24c00a4061 | ||
|
|
f38e4af226 | ||
|
|
62df6c39af | ||
|
|
d261458ecd | ||
|
|
7dfc7e4abc | ||
|
|
1bbb578afd | ||
|
|
7028cb4167 | ||
|
|
d4154e0349 | ||
|
|
b268755d51 | ||
|
|
a3aeef45cc | ||
|
|
aabe7756be | ||
|
|
4785cd959a | ||
|
|
b111076301 | ||
|
|
1dd613cb89 | ||
|
|
409399c609 | ||
|
|
43d5d66d34 | ||
|
|
f28f613f85 | ||
|
|
afe14ccbfa | ||
|
|
3674c0754e | ||
|
|
f2a3c27372 | ||
|
|
b0df3e62a8 | ||
|
|
6236749867 | ||
|
|
81ffa07439 | ||
|
|
265d287615 | ||
|
|
337e979a59 | ||
|
|
215818379b | ||
|
|
ac3449b0c8 | ||
|
|
e146418f65 | ||
|
|
a1f6823060 | ||
|
|
a6dbb09058 | ||
|
|
27701ef823 | ||
|
|
a286a1a6f7 | ||
|
|
a03b930339 | ||
|
|
6540bb32a6 | ||
|
|
bba088ef11 | ||
|
|
1fa09d9ede | ||
|
|
8b18cc2a94 | ||
|
|
dd69114573 | ||
|
|
e19f901330 | ||
|
|
d71444857e | ||
|
|
44bc7dc73d | ||
|
|
229adfb7c3 | ||
|
|
952f729b07 | ||
|
|
e652062f92 | ||
|
|
846753f343 | ||
|
|
07d4ed7e4c | ||
|
|
759ebea4eb | ||
|
|
132f09fab7 | ||
|
|
0d86288bd7 | ||
|
|
a75da49951 | ||
|
|
2407fecdae | ||
|
|
b12d1d866c | ||
|
|
6a50ab6b87 | ||
|
|
9d4cccd0f9 | ||
|
|
aefabaf774 | ||
|
|
b975830424 | ||
|
|
7123df3928 | ||
|
|
aaea6b97ad | ||
|
|
58653b5eae | ||
|
|
fb8ee02424 | ||
|
|
5a6817d5f8 | ||
|
|
4267c45db3 | ||
|
|
e39b25cd36 | ||
|
|
b057a90d49 | ||
|
|
38f0fa7bde | ||
|
|
1c81ec9248 | ||
|
|
9ff03680ba | ||
|
|
698392334f | ||
|
|
1e679bd789 | ||
|
|
9832599c9e | ||
|
|
bb8de51e5f | ||
|
|
91a4de4ca7 | ||
|
|
66e9d54eed | ||
|
|
8de6db15ac | ||
|
|
5954a0975f | ||
|
|
2e0eb88549 | ||
|
|
d6f9606e93 | ||
|
|
bd4a9473b0 | ||
|
|
a2c7b807e0 | ||
|
|
9eff7cd1d8 | ||
|
|
56cd47a159 |
@@ -225,13 +225,22 @@ runs:
|
|||||||
- name: Install gpuocelot dependencies (MacOS)
|
- name: Install gpuocelot dependencies (MacOS)
|
||||||
if: inputs.ocelot == 'true' && runner.os == 'macOS'
|
if: inputs.ocelot == 'true' && runner.os == 'macOS'
|
||||||
shell: bash
|
shell: bash
|
||||||
run: brew install --quiet cmake ninja llvm@15 zlib glew flex bison boost zstd ncurses
|
run: |
|
||||||
|
pkgs=(cmake ninja llvm@15 zlib glew flex bison [email protected] zstd ncurses)
|
||||||
|
for f in "${pkgs[@]}"; do
|
||||||
|
brew ls --versions "$f" >/dev/null 2>&1 || brew install --quiet "$f"
|
||||||
|
done
|
||||||
|
|
||||||
|
# Fix boost 1.85 for gpuocelot
|
||||||
|
ln -s /opt/homebrew/opt/[email protected] /opt/homebrew/opt/boost || true
|
||||||
|
ln -s /opt/homebrew/opt/boost/lib/libboost_atomic-mt.dylib /opt/homebrew/opt/boost/lib/libboost_atomic.dylib || true
|
||||||
|
ln -s /opt/homebrew/opt/boost/lib/libboost_thread-mt.dylib /opt/homebrew/opt/boost/lib/libboost_thread.dylib || true
|
||||||
- name: Cache gpuocelot
|
- name: Cache gpuocelot
|
||||||
if: inputs.ocelot == 'true'
|
if: inputs.ocelot == 'true'
|
||||||
id: cache-build
|
id: cache-build
|
||||||
uses: actions/cache@v4
|
uses: actions/cache@v4
|
||||||
env:
|
env:
|
||||||
cache-name: cache-gpuocelot-build
|
cache-name: cache-gpuocelot-build-1
|
||||||
with:
|
with:
|
||||||
path: ${{ github.workspace }}/gpuocelot/ocelot
|
path: ${{ github.workspace }}/gpuocelot/ocelot
|
||||||
key: ${{ runner.os }}-gpuocelot-b16039dc940dc6bc4ea0a98380495769ff35ed99-rebuild-${{ env.BUILD_CACHE_VERSION }}
|
key: ${{ runner.os }}-gpuocelot-b16039dc940dc6bc4ea0a98380495769ff35ed99-rebuild-${{ env.BUILD_CACHE_VERSION }}
|
||||||
@@ -244,7 +253,13 @@ runs:
|
|||||||
git checkout b16039dc940dc6bc4ea0a98380495769ff35ed99
|
git checkout b16039dc940dc6bc4ea0a98380495769ff35ed99
|
||||||
mkdir build
|
mkdir build
|
||||||
cd build
|
cd build
|
||||||
cmake .. -Wno-dev -G Ninja -DOCELOT_BUILD_TOOLS=OFF -DCMAKE_BUILD_ALWAYS=0 -DBUILD_TESTS_CUDA=OFF -DCMAKE_POLICY_VERSION_MINIMUM=3.5
|
|
||||||
|
CMAKE_ARGS="-Wno-dev -G Ninja -DOCELOT_BUILD_TOOLS=OFF -DCMAKE_BUILD_ALWAYS=0 -DBUILD_TESTS_CUDA=OFF -DCMAKE_POLICY_VERSION_MINIMUM=3.5"
|
||||||
|
if [[ "${{ runner.os }}" == "macOS" ]]; then
|
||||||
|
CMAKE_ARGS="$CMAKE_ARGS -DBoost_INCLUDE_DIR=$(brew --prefix boost)/include -DBoost_LIBRARY_DIR=$(brew --prefix boost)/lib"
|
||||||
|
fi
|
||||||
|
|
||||||
|
cmake .. $CMAKE_ARGS
|
||||||
ninja
|
ninja
|
||||||
- name: Install gpuocelot
|
- name: Install gpuocelot
|
||||||
if: inputs.ocelot == 'true'
|
if: inputs.ocelot == 'true'
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
name: Autogen
|
||||||
|
env:
|
||||||
|
# increment this when downloads substantially change to avoid the internet
|
||||||
|
DOWNLOAD_CACHE_VERSION: '12'
|
||||||
|
PYTHON_CACHE_VERSION: '3'
|
||||||
|
APT_CACHE_VERSION: '1'
|
||||||
|
BUILD_CACHE_VERSION: '1'
|
||||||
|
CAPTURE_PROCESS_REPLAY: 1
|
||||||
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
PYTHONPATH: ${{ github.workspace }}
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- master
|
||||||
|
pull_request:
|
||||||
|
paths:
|
||||||
|
- 'tinygrad/runtime/autogen/**/*'
|
||||||
|
workflow_dispatch:
|
||||||
|
paths:
|
||||||
|
- 'tinygrad/runtime/autogen/**/*'
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
autogen:
|
||||||
|
name: Autogen
|
||||||
|
runs-on: ubuntu-24.04
|
||||||
|
timeout-minutes: 15
|
||||||
|
steps:
|
||||||
|
- name: Checkout Code
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
- name: Setup Environment
|
||||||
|
uses: ./.github/actions/setup-tinygrad
|
||||||
|
with:
|
||||||
|
opencl: 'true'
|
||||||
|
amd: 'true'
|
||||||
|
cuda: 'true'
|
||||||
|
webgpu: 'true'
|
||||||
|
llvm: 'true'
|
||||||
|
- name: Install autogen support packages
|
||||||
|
run: sudo apt-get install -y --no-install-recommends llvm-14-dev libclang-14-dev
|
||||||
|
- name: Verify OpenCL autogen
|
||||||
|
run: |
|
||||||
|
cp tinygrad/runtime/autogen/opencl.py /tmp/opencl.py.bak
|
||||||
|
./autogen_stubs.sh opencl
|
||||||
|
diff /tmp/opencl.py.bak tinygrad/runtime/autogen/opencl.py
|
||||||
|
- name: Verify CUDA autogen
|
||||||
|
run: |
|
||||||
|
cp tinygrad/runtime/autogen/cuda.py /tmp/cuda.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/nv_gpu.py /tmp/nv_gpu.py.bak
|
||||||
|
./autogen_stubs.sh cuda
|
||||||
|
./autogen_stubs.sh nv
|
||||||
|
diff /tmp/cuda.py.bak tinygrad/runtime/autogen/cuda.py
|
||||||
|
diff /tmp/nv_gpu.py.bak tinygrad/runtime/autogen/nv_gpu.py
|
||||||
|
- name: Verify AMD autogen
|
||||||
|
run: |
|
||||||
|
cp tinygrad/runtime/autogen/hsa.py /tmp/hsa.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/kfd.py /tmp/kfd.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/comgr.py /tmp/comgr.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/amd_gpu.py /tmp/amd_gpu.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/sqtt.py /tmp/sqtt.py.bak
|
||||||
|
./autogen_stubs.sh hsa
|
||||||
|
./autogen_stubs.sh kfd
|
||||||
|
./autogen_stubs.sh comgr
|
||||||
|
./autogen_stubs.sh amd
|
||||||
|
./autogen_stubs.sh sqtt
|
||||||
|
diff /tmp/hsa.py.bak tinygrad/runtime/autogen/hsa.py
|
||||||
|
diff /tmp/kfd.py.bak tinygrad/runtime/autogen/kfd.py
|
||||||
|
diff /tmp/comgr.py.bak tinygrad/runtime/autogen/comgr.py
|
||||||
|
diff /tmp/amd_gpu.py.bak tinygrad/runtime/autogen/amd_gpu.py
|
||||||
|
diff /tmp/sqtt.py.bak tinygrad/runtime/autogen/sqtt.py
|
||||||
|
- name: Verify Linux autogen
|
||||||
|
run: |
|
||||||
|
cp tinygrad/runtime/autogen/libc.py /tmp/libc.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/io_uring.py /tmp/io_uring.py.bak
|
||||||
|
cp tinygrad/runtime/autogen/ib.py /tmp/ib.py.bak
|
||||||
|
./autogen_stubs.sh libc
|
||||||
|
./autogen_stubs.sh io_uring
|
||||||
|
./autogen_stubs.sh ib
|
||||||
|
diff /tmp/libc.py.bak tinygrad/runtime/autogen/libc.py
|
||||||
|
diff /tmp/io_uring.py.bak tinygrad/runtime/autogen/io_uring.py
|
||||||
|
diff /tmp/ib.py.bak tinygrad/runtime/autogen/ib.py
|
||||||
|
- name: Verify WebGPU autogen
|
||||||
|
run: |
|
||||||
|
cp tinygrad/runtime/autogen/webgpu.py /tmp/webgpu.py.bak
|
||||||
|
./autogen_stubs.sh webgpu
|
||||||
|
diff /tmp/webgpu.py.bak tinygrad/runtime/autogen/webgpu.py
|
||||||
|
- name: Verify LLVM autogen
|
||||||
|
run: |
|
||||||
|
cp tinygrad/runtime/autogen/llvm.py /tmp/llvm.py.bak
|
||||||
|
./autogen_stubs.sh llvm
|
||||||
|
diff /tmp/llvm.py.bak tinygrad/runtime/autogen/llvm.py
|
||||||
+105
-82
@@ -28,7 +28,7 @@ jobs:
|
|||||||
# since sudo is required for usbgpu on macos, move the cache to a new location, as some of the files are owned by root
|
# since sudo is required for usbgpu on macos, move the cache to a new location, as some of the files are owned by root
|
||||||
PYTHONPYCACHEPREFIX: /tmp/tiny_python_pycache
|
PYTHONPYCACHEPREFIX: /tmp/tiny_python_pycache
|
||||||
runs-on: [self-hosted, macOS]
|
runs-on: [self-hosted, macOS]
|
||||||
timeout-minutes: 20
|
timeout-minutes: 60
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -52,24 +52,28 @@ jobs:
|
|||||||
- name: reset process replay
|
- name: reset process replay
|
||||||
run: python3.11 test/external/process_replay/reset.py
|
run: python3.11 test/external/process_replay/reset.py
|
||||||
- name: Run Stable Diffusion
|
- name: Run Stable Diffusion
|
||||||
run: BENCHMARK_LOG=stable_diffusion JIT=1 python3.11 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing | tee sd.txt
|
run: BENCHMARK_LOG=stable_diffusion JIT=1 ASSERT_MIN_STEP_TIME=800 python3.11 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing | tee sd.txt
|
||||||
- name: Run Stable Diffusion without fp16
|
- name: Run Stable Diffusion without fp16
|
||||||
run: BENCHMARK_LOG=stable_diffusion_fp32 JIT=1 python3.11 examples/stable_diffusion.py --seed 0 --noshow --timing | tee sd_no_fp16.txt
|
run: BENCHMARK_LOG=stable_diffusion_fp32 JIT=1 ASSERT_MIN_STEP_TIME=900 python3.11 examples/stable_diffusion.py --seed 0 --noshow --timing | tee sd_no_fp16.txt
|
||||||
- name: Run Stable Diffusion v2
|
- name: Run Stable Diffusion v2
|
||||||
run: BENCHMARK_LOG=stable_diffusion_v2 JIT=1 python3.11 examples/sdv2.py --fp16 --seed 0 --noshow --timing | tee sdv2.txt
|
# TODO: very slow step time
|
||||||
|
run: BENCHMARK_LOG=stable_diffusion_v2 JIT=1 ASSERT_MIN_STEP_TIME=10000 python3.11 examples/sdv2.py --fp16 --seed 0 --noshow --timing | tee sdv2.txt
|
||||||
# process replay can't capture this, the graph is too large
|
# process replay can't capture this, the graph is too large
|
||||||
- name: Run SDXL
|
# TODO: too slow
|
||||||
run: BENCHMARK_LOG=stable_diffusion_xl CAPTURE_PROCESS_REPLAY=0 JIT=1 python3.11 examples/sdxl.py --seed 0 --noshow --timing | tee sdxl.txt
|
# - name: Run SDXL
|
||||||
|
# run: BENCHMARK_LOG=stable_diffusion_xl ASSERT_MIN_STEP_TIME=5000 CAPTURE_PROCESS_REPLAY=0 JIT=1 python3.11 examples/sdxl.py --seed 0 --noshow --timing | tee sdxl.txt
|
||||||
- name: Run model inference benchmark
|
- name: Run model inference benchmark
|
||||||
run: METAL=1 python3.11 test/external/external_model_benchmark.py
|
run: METAL=1 python3.11 test/external/external_model_benchmark.py
|
||||||
- name: Test speed vs torch
|
- name: Test speed vs torch
|
||||||
run: BIG=2 MPS=1 python3.11 test/speed/external_test_speed_v_torch.py | tee torch_speed.txt
|
run: BIG=2 MPS=1 python3.11 test/speed/external_test_speed_v_torch.py | tee torch_speed.txt
|
||||||
- name: Test tensor cores
|
- name: Test tensor cores
|
||||||
run: METAL=1 python3.11 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
run: METAL=1 python3.11 test/opt/test_tensor_cores.py
|
||||||
- name: Test AMX tensor cores
|
- name: Test AMX tensor cores
|
||||||
run: |
|
run: |
|
||||||
DEBUG=2 CPU=1 AMX=1 python3.11 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops TestFloat4.test_float4_multidim_amx TestFloat4.test_float4_multidim_unaligned_load_amx
|
DEBUG=2 CPU=1 CPU_LLVM=0 AMX=1 python3.11 test/opt/test_tensor_cores.py
|
||||||
DEBUG=2 LLVM=1 AMX=1 python3.11 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops TestFloat4.test_float4_multidim_amx TestFloat4.test_float4_multidim_unaligned_load_amx
|
DEBUG=2 CPU=1 CPU_LLVM=1 AMX=1 python3.11 test/opt/test_tensor_cores.py
|
||||||
|
DEBUG=2 CPU=1 CPU_LLVM=0 AMX=1 python3.11 test/opt/test_gen_float4.py TestFloat4.test_float4_multidim_amx TestFloat4.test_float4_multidim_unaligned_load_amx
|
||||||
|
DEBUG=2 CPU=1 CPU_LLVM=1 AMX=1 python3.11 test/opt/test_gen_float4.py TestFloat4.test_float4_multidim_amx TestFloat4.test_float4_multidim_unaligned_load_amx
|
||||||
- name: Run Tensor Core GEMM (float)
|
- name: Run Tensor Core GEMM (float)
|
||||||
run: DEBUG=2 SHOULD_USE_TC=1 python3.11 extra/gemm/simple_matmul.py | tee matmul.txt
|
run: DEBUG=2 SHOULD_USE_TC=1 python3.11 extra/gemm/simple_matmul.py | tee matmul.txt
|
||||||
- name: Run Tensor Core GEMM (half)
|
- name: Run Tensor Core GEMM (half)
|
||||||
@@ -97,7 +101,7 @@ jobs:
|
|||||||
- name: Run GPT2
|
- name: Run GPT2
|
||||||
run: |
|
run: |
|
||||||
BENCHMARK_LOG=gpt2_nojit JIT=0 python3.11 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_unjitted.txt
|
BENCHMARK_LOG=gpt2_nojit JIT=0 python3.11 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_unjitted.txt
|
||||||
BENCHMARK_LOG=gpt2 JIT=1 python3.11 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_jitted.txt
|
BENCHMARK_LOG=gpt2 JIT=1 ASSERT_MIN_STEP_TIME=13 python3.11 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_jitted.txt
|
||||||
- name: Run GPT2 w HALF
|
- name: Run GPT2 w HALF
|
||||||
run: BENCHMARK_LOG=gpt2_half HALF=1 python3.11 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half.txt
|
run: BENCHMARK_LOG=gpt2_half HALF=1 python3.11 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half.txt
|
||||||
- name: Run GPT2 w HALF/BEAM
|
- name: Run GPT2 w HALF/BEAM
|
||||||
@@ -106,22 +110,27 @@ jobs:
|
|||||||
run: BENCHMARK_LOG=olmoe python3.11 examples/olmoe.py
|
run: BENCHMARK_LOG=olmoe python3.11 examples/olmoe.py
|
||||||
- name: Train MNIST
|
- name: Train MNIST
|
||||||
run: time PYTHONPATH=. TARGET_EVAL_ACC_PCT=96.0 python3.11 examples/beautiful_mnist.py | tee beautiful_mnist.txt
|
run: time PYTHONPATH=. TARGET_EVAL_ACC_PCT=96.0 python3.11 examples/beautiful_mnist.py | tee beautiful_mnist.txt
|
||||||
- name: Run 10 CIFAR training steps
|
|
||||||
run: BENCHMARK_LOG=cifar_10steps JIT=1 STEPS=10 python3.11 examples/hlb_cifar10.py | tee train_cifar.txt
|
# NOTE: this is failing in CI. it is not failing on my machine and I don't really have a way to debug it
|
||||||
- name: Run 10 CIFAR training steps w HALF
|
# the error is "RuntimeError: Internal Error (0000000e:Internal Error)"
|
||||||
run: BENCHMARK_LOG=cifar_10steps_half JIT=2 STEPS=10 DEFAULT_FLOAT=HALF python3.11 examples/hlb_cifar10.py | tee train_cifar_half.txt
|
#- name: Run 10 CIFAR training steps
|
||||||
|
# run: BENCHMARK_LOG=cifar_10steps JIT=1 ASSERT_MIN_STEP_TIME=3000 STEPS=10 python3.11 examples/hlb_cifar10.py | tee train_cifar.txt
|
||||||
|
#- name: Run 10 CIFAR training steps w HALF
|
||||||
|
# run: BENCHMARK_LOG=cifar_10steps_half JIT=2 ASSERT_MIN_STEP_TIME=3000 STEPS=10 DEFAULT_FLOAT=HALF python3.11 examples/hlb_cifar10.py | tee train_cifar_half.txt
|
||||||
|
|
||||||
#- name: Run 10 CIFAR training steps w BF16
|
#- name: Run 10 CIFAR training steps w BF16
|
||||||
# run: STEPS=10 DEFAULT_FLOAT=BFLOAT16 python3.11 examples/hlb_cifar10.py | tee train_cifar_bf16.txt
|
# run: STEPS=10 DEFAULT_FLOAT=BFLOAT16 python3.11 examples/hlb_cifar10.py | tee train_cifar_bf16.txt
|
||||||
- name: Run 10 CIFAR training steps w winograd
|
# TODO: too slow
|
||||||
run: BENCHMARK_LOG=cifar_10steps_wino JIT=1 WINO=1 STEPS=10 python3.11 examples/hlb_cifar10.py | tee train_cifar_wino.txt
|
# - name: Run 10 CIFAR training steps w winograd
|
||||||
|
# run: BENCHMARK_LOG=cifar_10steps_wino JIT=1 ASSERT_MIN_STEP_TIME=150 WINO=1 STEPS=10 python3.11 examples/hlb_cifar10.py | tee train_cifar_wino.txt
|
||||||
- name: UsbGPU boot time
|
- name: UsbGPU boot time
|
||||||
run: sudo -E PYTHONPATH=. DEBUG=2 AM_RESET=1 AMD=1 AMD_IFACE=USB time python3.11 test/test_tiny.py TestTiny.test_plus
|
run: sudo -E PYTHONPATH=. DEBUG=2 AM_RESET=1 AMD=1 AMD_IFACE=USB time python3.11 test/test_tiny.py TestTiny.test_plus
|
||||||
- name: UsbGPU tiny tests
|
- name: UsbGPU tiny tests
|
||||||
run: sudo -E PYTHONPATH=. AMD=1 AMD_IFACE=USB python3.11 test/test_tiny.py
|
run: sudo -E PYTHONPATH=. AMD=1 AMD_IFACE=USB python3.11 test/test_tiny.py
|
||||||
- name: UsbGPU copy speeds
|
- name: UsbGPU copy speeds
|
||||||
run: sudo -E PYTHONPATH=. AMD=1 AMD_IFACE=USB python3.11 test/external/external_test_usb_asm24.py TestDevCopySpeeds
|
run: sudo -E PYTHONPATH=. AMD=1 AMD_IFACE=USB python3.11 test/external/external_test_usb_asm24.py TestDevCopySpeeds
|
||||||
- name: UsbGPU openpilot test
|
#- name: UsbGPU openpilot test
|
||||||
run: sudo -E PYTHONPATH=. AMD=1 AMD_IFACE=USB NOLOCALS=0 IMAGE=0 GRAPH_ONE_KERNEL=1 python3.11 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/9118973ed03c1ae1d40cf69a29507ec2cc78efd7/selfdrive/modeld/models/supercombo.onnx
|
# run: sudo -E PYTHONPATH=. AMD=1 AMD_IFACE=USB NOLOCALS=0 IMAGE=0 GRAPH_ONE_KERNEL=1 python3.11 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/9118973ed03c1ae1d40cf69a29507ec2cc78efd7/selfdrive/modeld/models/supercombo.onnx
|
||||||
- uses: actions/upload-artifact@v4
|
- uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: Speed (Mac)
|
name: Speed (Mac)
|
||||||
@@ -158,7 +167,7 @@ jobs:
|
|||||||
testnvidiabenchmark:
|
testnvidiabenchmark:
|
||||||
name: tinybox green Benchmark
|
name: tinybox green Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinyboxgreen]
|
runs-on: [self-hosted, Linux, tinyboxgreen]
|
||||||
timeout-minutes: 30
|
timeout-minutes: 60
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -194,15 +203,15 @@ jobs:
|
|||||||
run: NV=1 python test/external/external_benchmark_multitensor_allreduce.py
|
run: NV=1 python test/external/external_benchmark_multitensor_allreduce.py
|
||||||
- name: Test tensor cores
|
- name: Test tensor cores
|
||||||
run: |
|
run: |
|
||||||
NV=1 ALLOW_TF32=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
NV=1 ALLOW_TF32=1 python3 test/opt/test_tensor_cores.py
|
||||||
PTX=1 ALLOW_TF32=1 NV=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
NV=1 NV_PTX=1 ALLOW_TF32=1 python3 test/opt/test_tensor_cores.py
|
||||||
- name: Run Tensor Core GEMM (CUDA)
|
- name: Run Tensor Core GEMM (CUDA)
|
||||||
run: |
|
run: |
|
||||||
CUDA=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul.txt
|
CUDA=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul.txt
|
||||||
CUDA=1 SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul_bfloat16.txt
|
CUDA=1 SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul_bfloat16.txt
|
||||||
CUDA=1 SHOULD_USE_TC=1 ALLOW_TF32=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py | tee matmul_tf32.txt
|
CUDA=1 SHOULD_USE_TC=1 ALLOW_TF32=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py | tee matmul_tf32.txt
|
||||||
- name: Run Tensor Core GEMM (PTX)
|
- name: Run Tensor Core GEMM (PTX)
|
||||||
run: NV=1 PTX=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul_ptx.txt
|
run: NV=1 NV_PTX=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul_ptx.txt
|
||||||
- name: Run Tensor Core GEMM (NV)
|
- name: Run Tensor Core GEMM (NV)
|
||||||
run: NV=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul_nv.txt
|
run: NV=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 python3 extra/gemm/simple_matmul.py | tee matmul_nv.txt
|
||||||
- name: Test NV=1
|
- name: Test NV=1
|
||||||
@@ -211,8 +220,9 @@ jobs:
|
|||||||
run: DEBUG=2 CUDA=1 python -m pytest -rA test/test_tiny.py
|
run: DEBUG=2 CUDA=1 python -m pytest -rA test/test_tiny.py
|
||||||
- name: Run Stable Diffusion
|
- name: Run Stable Diffusion
|
||||||
run: BENCHMARK_LOG=stable_diffusion NV=1 python3 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing | tee sd.txt
|
run: BENCHMARK_LOG=stable_diffusion NV=1 python3 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing | tee sd.txt
|
||||||
- name: Run SDXL
|
# TODO: too slow
|
||||||
run: BENCHMARK_LOG=stable_diffusion_xl CAPTURE_PROCESS_REPLAY=0 NV=1 CAPTURE_PROCESS_REPLAY=0 python3 examples/sdxl.py --seed 0 --noshow --timing | tee sdxl.txt
|
# - name: Run SDXL
|
||||||
|
# run: BENCHMARK_LOG=stable_diffusion_xl ASSERT_MIN_STEP_TIME=2000 CAPTURE_PROCESS_REPLAY=0 NV=1 CAPTURE_PROCESS_REPLAY=0 python3 examples/sdxl.py --seed 0 --noshow --timing | tee sdxl.txt
|
||||||
- name: Run LLaMA
|
- name: Run LLaMA
|
||||||
run: |
|
run: |
|
||||||
BENCHMARK_LOG=llama_nojit NV=1 JIT=0 python3 examples/llama.py --gen 1 --prompt "Hello." --count 10 --temperature 0 --timing | tee llama_unjitted.txt
|
BENCHMARK_LOG=llama_nojit NV=1 JIT=0 python3 examples/llama.py --gen 1 --prompt "Hello." --count 10 --temperature 0 --timing | tee llama_unjitted.txt
|
||||||
@@ -236,9 +246,9 @@ jobs:
|
|||||||
- name: Run GPT2
|
- name: Run GPT2
|
||||||
run: |
|
run: |
|
||||||
BENCHMARK_LOG=gpt2_nojit NV=1 JIT=0 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_unjitted.txt
|
BENCHMARK_LOG=gpt2_nojit NV=1 JIT=0 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_unjitted.txt
|
||||||
BENCHMARK_LOG=gpt2 NV=1 JIT=1 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_jitted.txt
|
BENCHMARK_LOG=gpt2 NV=1 JIT=1 ASSERT_MIN_STEP_TIME=4 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_jitted.txt
|
||||||
- name: Run GPT2 w HALF
|
- name: Run GPT2 w HALF
|
||||||
run: BENCHMARK_LOG=gpt2_half NV=1 HALF=1 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half.txt
|
run: BENCHMARK_LOG=gpt2_half NV=1 HALF=1 ASSERT_MIN_STEP_TIME=6 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half.txt
|
||||||
- name: Run GPT2 w HALF/BEAM
|
- name: Run GPT2 w HALF/BEAM
|
||||||
run: BENCHMARK_LOG=gpt2_half_beam NV=1 HALF=1 JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half_beam.txt
|
run: BENCHMARK_LOG=gpt2_half_beam NV=1 HALF=1 JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half_beam.txt
|
||||||
- uses: actions/upload-artifact@v4
|
- uses: actions/upload-artifact@v4
|
||||||
@@ -272,7 +282,7 @@ jobs:
|
|||||||
testmorenvidiabenchmark:
|
testmorenvidiabenchmark:
|
||||||
name: tinybox green Training Benchmark
|
name: tinybox green Training Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinyboxgreen]
|
runs-on: [self-hosted, Linux, tinyboxgreen]
|
||||||
timeout-minutes: 20
|
timeout-minutes: 60
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -297,30 +307,33 @@ jobs:
|
|||||||
rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
|
rm -f /tmp/staging.db /tmp/staging.db-shm /tmp/staging.db-wal
|
||||||
- name: reset process replay
|
- name: reset process replay
|
||||||
run: test/external/process_replay/reset.py
|
run: test/external/process_replay/reset.py
|
||||||
- name: Fuzz Padded Tensor Core GEMM (NV)
|
# TODO: too slow
|
||||||
run: NV=1 M_START=12 M_STOP=20 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 python3 ./extra/gemm/fuzz_matmul.py
|
# - name: Fuzz Padded Tensor Core GEMM (NV)
|
||||||
- name: Fuzz Padded Tensor Core GEMM (PTX)
|
# run: NV=1 M_START=12 M_STOP=20 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 python3 ./extra/gemm/fuzz_matmul.py
|
||||||
run: NV=1 PTX=1 M_START=12 M_STOP=20 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 python3 ./extra/gemm/fuzz_matmul.py
|
# TODO: too slow
|
||||||
|
# - name: Fuzz Padded Tensor Core GEMM (PTX)
|
||||||
|
# run: NV=1 NV_PTX=1 M_START=12 M_STOP=20 M_STEP=1 N_START=6 N_STOP=10 N_STEP=1 K_START=28 K_STOP=36 K_STEP=1 HALF=1 TC_OPT=2 python3 ./extra/gemm/fuzz_matmul.py
|
||||||
- name: Train MNIST
|
- name: Train MNIST
|
||||||
run: time PYTHONPATH=. NV=1 TARGET_EVAL_ACC_PCT=96.0 python3 examples/beautiful_mnist.py | tee beautiful_mnist.txt
|
run: time PYTHONPATH=. NV=1 TARGET_EVAL_ACC_PCT=96.0 python3 examples/beautiful_mnist.py | tee beautiful_mnist.txt
|
||||||
- name: Run 10 CIFAR training steps
|
- name: Run 10 CIFAR training steps
|
||||||
run: BENCHMARK_LOG=cifar_10steps NV=1 STEPS=10 python3 examples/hlb_cifar10.py | tee train_cifar.txt
|
run: BENCHMARK_LOG=cifar_10steps ASSERT_MIN_STEP_TIME=270 NV=1 STEPS=10 python3 examples/hlb_cifar10.py | tee train_cifar.txt
|
||||||
- name: Run 10 CIFAR training steps w HALF
|
- name: Run 10 CIFAR training steps w HALF
|
||||||
run: BENCHMARK_LOG=cifar_10steps_half NV=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_half.txt
|
run: BENCHMARK_LOG=cifar_10steps_half ASSERT_MIN_STEP_TIME=310 NV=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_half.txt
|
||||||
- name: Run 10 CIFAR training steps w BF16
|
- name: Run 10 CIFAR training steps w BF16
|
||||||
run: BENCHMARK_LOG=cifar_10steps_bf16 NV=1 STEPS=10 DEFAULT_FLOAT=BFLOAT16 python3 examples/hlb_cifar10.py | tee train_cifar_bf16.txt
|
run: BENCHMARK_LOG=cifar_10steps_bf16 ASSERT_MIN_STEP_TIME=310 NV=1 STEPS=10 DEFAULT_FLOAT=BFLOAT16 python3 examples/hlb_cifar10.py | tee train_cifar_bf16.txt
|
||||||
- name: Run 10 CIFAR training steps w winograd
|
# TODO: too slow
|
||||||
run: BENCHMARK_LOG=cifar_10steps_half_wino NV=1 CAPTURE_PROCESS_REPLAY=0 WINO=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_wino.txt
|
# - name: Run 10 CIFAR training steps w winograd
|
||||||
|
# run: BENCHMARK_LOG=cifar_10steps_half_wino ASSERT_MIN_STEP_TIME=350 NV=1 CAPTURE_PROCESS_REPLAY=0 WINO=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_wino.txt
|
||||||
- name: Run full CIFAR training w 1 GPU
|
- name: Run full CIFAR training w 1 GPU
|
||||||
run: time BENCHMARK_LOG=cifar NV=1 DEFAULT_FLOAT=HALF LATEWINO=1 STEPS=1000 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee train_cifar_one_gpu.txt
|
run: time BENCHMARK_LOG=cifar NV=1 DEFAULT_FLOAT=HALF STEPS=1000 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee train_cifar_one_gpu.txt
|
||||||
- name: Run full CIFAR training steps w 6 GPUS
|
- name: Run full CIFAR training steps w 6 GPUS
|
||||||
run: time BENCHMARK_LOG=cifar_6gpu CAPTURE_PROCESS_REPLAY=0 NV=1 DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee train_cifar_six_gpu.txt
|
run: time BENCHMARK_LOG=cifar_6gpu CAPTURE_PROCESS_REPLAY=0 NV=1 DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee train_cifar_six_gpu.txt
|
||||||
- name: Run MLPerf resnet eval on training data
|
- name: Run MLPerf resnet eval on training data
|
||||||
run: time BENCHMARK_LOG=resnet_eval NV=1 MODEL=resnet python3 examples/mlperf/model_eval.py
|
run: time BENCHMARK_LOG=resnet_eval NV=1 MODEL=resnet python3 examples/mlperf/model_eval.py
|
||||||
- name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
#- name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
||||||
run: BENCHMARK_LOG=resnet_10steps NV=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet_one_gpu.txt
|
# run: BENCHMARK_LOG=resnet_10steps NV=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet_one_gpu.txt
|
||||||
- name: Run 10 MLPerf ResNet50 training steps (6 gpu)
|
#- name: Run 10 MLPerf ResNet50 training steps (6 gpu)
|
||||||
run: BENCHMARK_LOG=resnet_10steps_6gpu NV=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=1536 GPUS=6 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet.txt
|
# run: BENCHMARK_LOG=resnet_10steps_6gpu NV=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=1536 GPUS=6 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet.txt
|
||||||
- name: Run 10 MLPerf Bert training steps (6 gpu)
|
- name: Run 10 MLPerf Bert training steps (6 gpu)
|
||||||
# TODO: remove BERT_LAYERS once scheduler is fast
|
# TODO: remove BERT_LAYERS once scheduler is fast
|
||||||
run: BENCHMARK_LOG=bert_10steps_6gpu NV=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=6 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py | tee train_bert.txt
|
run: BENCHMARK_LOG=bert_10steps_6gpu NV=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=6 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py | tee train_bert.txt
|
||||||
@@ -344,7 +357,7 @@ jobs:
|
|||||||
testamdbenchmark:
|
testamdbenchmark:
|
||||||
name: tinybox red Benchmark
|
name: tinybox red Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinybox]
|
runs-on: [self-hosted, Linux, tinybox]
|
||||||
timeout-minutes: 20
|
timeout-minutes: 60
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -394,8 +407,8 @@ jobs:
|
|||||||
run: AMD=1 IGNORE_BEAM_CACHE=1 BEAM_DEBUG=1 DEBUG=1 python -m pytest -rA test/external/speed_v_theoretical.py --durations=20
|
run: AMD=1 IGNORE_BEAM_CACHE=1 BEAM_DEBUG=1 DEBUG=1 python -m pytest -rA test/external/speed_v_theoretical.py --durations=20
|
||||||
- name: Test tensor cores
|
- name: Test tensor cores
|
||||||
run: |
|
run: |
|
||||||
AMD=1 AMD_LLVM=0 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded_amd TestLinearizer.test_tensor_cores_padded_uops
|
AMD=1 AMD_LLVM=0 python3 test/opt/test_tensor_cores.py
|
||||||
AMD=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded_amd TestLinearizer.test_tensor_cores_padded_uops
|
AMD=1 AMD_LLVM=1 python3 test/opt/test_tensor_cores.py
|
||||||
AMD=1 SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
|
AMD=1 SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
|
||||||
- name: Run Tensor Core GEMM (AMD)
|
- name: Run Tensor Core GEMM (AMD)
|
||||||
run: AMD=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py | tee matmul_amd.txt
|
run: AMD=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py | tee matmul_amd.txt
|
||||||
@@ -413,9 +426,10 @@ jobs:
|
|||||||
- name: Test AM warm start time
|
- name: Test AM warm start time
|
||||||
run: time AMD=1 python3 test/test_tiny.py TestTiny.test_plus
|
run: time AMD=1 python3 test/test_tiny.py TestTiny.test_plus
|
||||||
- name: Run Stable Diffusion
|
- name: Run Stable Diffusion
|
||||||
run: BENCHMARK_LOG=stable_diffusion AMD=1 python3 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing | tee sd.txt
|
run: BENCHMARK_LOG=stable_diffusion ASSERT_MIN_STEP_TIME=550 AMD=1 python3 examples/stable_diffusion.py --fp16 --seed 0 --noshow --timing | tee sd.txt
|
||||||
- name: Run SDXL
|
# TODO: too slow
|
||||||
run: BENCHMARK_LOG=stable_diffusion_xl CAPTURE_PROCESS_REPLAY=0 AMD=1 python3 examples/sdxl.py --seed 0 --noshow --timing | tee sdxl.txt
|
# - name: Run SDXL
|
||||||
|
# run: BENCHMARK_LOG=stable_diffusion_xl ASSERT_MIN_STEP_TIME=3200 CAPTURE_PROCESS_REPLAY=0 AMD=1 python3 examples/sdxl.py --seed 0 --noshow --timing | tee sdxl.txt
|
||||||
- name: Run LLaMA 7B
|
- name: Run LLaMA 7B
|
||||||
run: |
|
run: |
|
||||||
BENCHMARK_LOG=llama_nojit AMD=1 JIT=0 python3 examples/llama.py --gen 1 --prompt "Hello." --count 10 --temperature 0 --timing | tee llama_unjitted.txt
|
BENCHMARK_LOG=llama_nojit AMD=1 JIT=0 python3 examples/llama.py --gen 1 --prompt "Hello." --count 10 --temperature 0 --timing | tee llama_unjitted.txt
|
||||||
@@ -441,9 +455,9 @@ jobs:
|
|||||||
- name: Run GPT2
|
- name: Run GPT2
|
||||||
run: |
|
run: |
|
||||||
BENCHMARK_LOG=gpt2_nojit AMD=1 JIT=0 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_unjitted.txt
|
BENCHMARK_LOG=gpt2_nojit AMD=1 JIT=0 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_unjitted.txt
|
||||||
BENCHMARK_LOG=gpt2 AMD=1 JIT=1 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_jitted.txt
|
BENCHMARK_LOG=gpt2 AMD=1 JIT=1 ASSERT_MIN_STEP_TIME=5 python3 examples/gpt2.py --prompt "Hello." --count 10 --temperature 0 --timing | tee gpt2_jitted.txt
|
||||||
- name: Run GPT2 w HALF
|
- name: Run GPT2 w HALF
|
||||||
run: BENCHMARK_LOG=gpt2_half AMD=1 HALF=1 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half.txt
|
run: BENCHMARK_LOG=gpt2_half AMD=1 HALF=1 ASSERT_MIN_STEP_TIME=5 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half.txt
|
||||||
- name: Run GPT2 w HALF/BEAM
|
- name: Run GPT2 w HALF/BEAM
|
||||||
run: BENCHMARK_LOG=gpt2_half_beam AMD=1 HALF=1 JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half_beam.txt
|
run: BENCHMARK_LOG=gpt2_half_beam AMD=1 HALF=1 JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 examples/gpt2.py --count 10 --temperature 0 --timing | tee gpt2_half_beam.txt
|
||||||
- uses: actions/upload-artifact@v4
|
- uses: actions/upload-artifact@v4
|
||||||
@@ -474,7 +488,7 @@ jobs:
|
|||||||
testmoreamdbenchmark:
|
testmoreamdbenchmark:
|
||||||
name: tinybox red Training Benchmark
|
name: tinybox red Training Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinybox]
|
runs-on: [self-hosted, Linux, tinybox]
|
||||||
timeout-minutes: 30
|
timeout-minutes: 60
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -506,19 +520,20 @@ jobs:
|
|||||||
- name: Train MNIST
|
- name: Train MNIST
|
||||||
run: time PYTHONPATH=. AMD=1 TARGET_EVAL_ACC_PCT=96.0 python3 examples/beautiful_mnist.py | tee beautiful_mnist.txt
|
run: time PYTHONPATH=. AMD=1 TARGET_EVAL_ACC_PCT=96.0 python3 examples/beautiful_mnist.py | tee beautiful_mnist.txt
|
||||||
- name: Run 10 CIFAR training steps
|
- name: Run 10 CIFAR training steps
|
||||||
run: BENCHMARK_LOG=cifar_10steps AMD=1 STEPS=10 python3 examples/hlb_cifar10.py | tee train_cifar.txt
|
run: BENCHMARK_LOG=cifar_10steps ASSERT_MIN_STEP_TIME=330 AMD=1 STEPS=10 python3 examples/hlb_cifar10.py | tee train_cifar.txt
|
||||||
- name: Run 10 CIFAR training steps w HALF
|
- name: Run 10 CIFAR training steps w HALF
|
||||||
run: BENCHMARK_LOG=cifar_10steps_half AMD=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_half.txt
|
run: BENCHMARK_LOG=cifar_10steps_half ASSERT_MIN_STEP_TIME=330 AMD=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_half.txt
|
||||||
- name: Run 10 CIFAR training steps w BF16
|
# - name: Run 10 CIFAR training steps w BF16
|
||||||
run: BENCHMARK_LOG=cifar_10steps_bf16 AMD=1 STEPS=10 DEFAULT_FLOAT=BFLOAT16 python3 examples/hlb_cifar10.py | tee train_cifar_bf16.txt
|
# run: BENCHMARK_LOG=cifar_10steps_bf16 ASSERT_MIN_STEP_TIME=288 AMD=1 STEPS=10 DEFAULT_FLOAT=BFLOAT16 python3 examples/hlb_cifar10.py | tee train_cifar_bf16.txt
|
||||||
- name: Run 10 CIFAR training steps w winograd
|
# TODO: too slow
|
||||||
run: BENCHMARK_LOG=cifar_10steps_half_wino AMD=1 WINO=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_wino.txt
|
# - name: Run 10 CIFAR training steps w winograd
|
||||||
|
# run: BENCHMARK_LOG=cifar_10steps_half_wino ASSERT_MIN_STEP_TIME=66 AMD=1 WINO=1 STEPS=10 DEFAULT_FLOAT=HALF python3 examples/hlb_cifar10.py | tee train_cifar_wino.txt
|
||||||
- name: Run full CIFAR training w 1 GPU
|
- name: Run full CIFAR training w 1 GPU
|
||||||
run: time BENCHMARK_LOG=cifar AMD=1 DEFAULT_FLOAT=HALF LATEWINO=1 STEPS=1000 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee train_cifar_one_gpu.txt
|
run: time BENCHMARK_LOG=cifar AMD=1 DEFAULT_FLOAT=HALF STEPS=1000 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee train_cifar_one_gpu.txt
|
||||||
- name: Run full CIFAR training steps w 6 GPUS
|
#- name: Run full CIFAR training steps w 6 GPUS
|
||||||
run: time BENCHMARK_LOG=cifar_6gpu AMD=1 DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee train_cifar_six_gpu.txt
|
# run: time BENCHMARK_LOG=cifar_6gpu AMD=1 DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee train_cifar_six_gpu.txt
|
||||||
- name: Run full CIFAR training steps w 6 GPUS (REMOTE)
|
#- name: Run full CIFAR training steps w 6 GPUS (REMOTE)
|
||||||
run: time BENCHMARK_LOG=cifar_6gpu_remote REMOTE=1 REMOTEDEV=AMD DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee train_cifar_six_gpu_remote.txt
|
# run: time BENCHMARK_LOG=cifar_6gpu_remote REMOTE=1 REMOTEDEV=AMD DEFAULT_FLOAT=HALF STEPS=350 BS=1536 GPUS=6 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee train_cifar_six_gpu_remote.txt
|
||||||
- uses: actions/upload-artifact@v4
|
- uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: Speed (AMD Training)
|
name: Speed (AMD Training)
|
||||||
@@ -537,7 +552,7 @@ jobs:
|
|||||||
testmlperfamdbenchmark:
|
testmlperfamdbenchmark:
|
||||||
name: tinybox red MLPerf Benchmark
|
name: tinybox red MLPerf Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinybox]
|
runs-on: [self-hosted, Linux, tinybox]
|
||||||
timeout-minutes: 30
|
timeout-minutes: 60
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -568,10 +583,10 @@ jobs:
|
|||||||
run: test/external/process_replay/reset.py
|
run: test/external/process_replay/reset.py
|
||||||
- name: Run MLPerf resnet eval
|
- name: Run MLPerf resnet eval
|
||||||
run: time BENCHMARK_LOG=resnet_eval AMD=1 MODEL=resnet python3 examples/mlperf/model_eval.py
|
run: time BENCHMARK_LOG=resnet_eval AMD=1 MODEL=resnet python3 examples/mlperf/model_eval.py
|
||||||
- name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
#- name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
||||||
run: BENCHMARK_LOG=resnet_10steps AMD=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet_one_gpu.txt
|
# run: BENCHMARK_LOG=resnet_10steps AMD=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet_one_gpu.txt
|
||||||
- name: Run 10 MLPerf ResNet50 training steps (6 gpu)
|
#- name: Run 10 MLPerf ResNet50 training steps (6 gpu)
|
||||||
run: BENCHMARK_LOG=resnet_10steps_6gpu AMD=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=1536 GPUS=6 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet.txt
|
# run: BENCHMARK_LOG=resnet_10steps_6gpu AMD=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=1536 GPUS=6 MODEL=resnet python3 examples/mlperf/model_train.py | tee train_resnet.txt
|
||||||
- name: Run 10 MLPerf Bert training steps (6 gpu)
|
- name: Run 10 MLPerf Bert training steps (6 gpu)
|
||||||
# TODO: remove BERT_LAYERS once scheduler is fast
|
# TODO: remove BERT_LAYERS once scheduler is fast
|
||||||
run: BENCHMARK_LOG=bert_10steps_6gpu AMD=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=6 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py | tee train_bert.txt
|
run: BENCHMARK_LOG=bert_10steps_6gpu AMD=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=6 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py | tee train_bert.txt
|
||||||
@@ -610,21 +625,21 @@ jobs:
|
|||||||
- name: benchmark openpilot 0.9.9 dmonitoring
|
- name: benchmark openpilot 0.9.9 dmonitoring
|
||||||
run: BENCHMARK_LOG=openpilot_0_9_9_dmonitoring PYTHONPATH=. NOLOCALS=1 FLOAT16=1 IMAGE=2 QCOM=1 taskset -c 4-7 python3 test/external/external_benchmark_openpilot.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/dmonitoring_model.onnx
|
run: BENCHMARK_LOG=openpilot_0_9_9_dmonitoring PYTHONPATH=. NOLOCALS=1 FLOAT16=1 IMAGE=2 QCOM=1 taskset -c 4-7 python3 test/external/external_benchmark_openpilot.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/dmonitoring_model.onnx
|
||||||
- name: openpilot compile3 0.9.9 driving_vision
|
- name: openpilot compile3 0.9.9 driving_vision
|
||||||
run: PYTHONPATH="." QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/driving_vision.onnx
|
run: PYTHONPATH="." ASSERT_MIN_STEP_TIME=18 QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/driving_vision.onnx
|
||||||
- name: openpilot compile3 0.9.9 driving_policy
|
- name: openpilot compile3 0.9.9 driving_policy
|
||||||
run: PYTHONPATH="." QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/driving_policy.onnx
|
run: PYTHONPATH="." ASSERT_MIN_STEP_TIME=7 QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/driving_policy.onnx
|
||||||
- name: openpilot compile3 0.9.9 dmonitoring
|
- name: openpilot compile3 0.9.9 dmonitoring
|
||||||
run: PYTHONPATH="." QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/dmonitoring_model.onnx
|
run: PYTHONPATH="." ASSERT_MIN_STEP_TIME=12 QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.9/selfdrive/modeld/models/dmonitoring_model.onnx
|
||||||
- name: openpilot compile3 Space Lab policy + vision
|
- name: openpilot compile3 Space Lab policy + vision
|
||||||
run: |
|
run: |
|
||||||
PYTHONPATH="." QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://gitlab.com/commaai/openpilot-lfs.git/gitlab-lfs/objects/22aec22a10ce09384d4a4af2a0bbff08d54af7e0c888503508f356fae4ff0e29
|
PYTHONPATH="." ASSERT_MIN_STEP_TIME=4 QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://gitlab.com/commaai/openpilot-lfs.git/gitlab-lfs/objects/22aec22a10ce09384d4a4af2a0bbff08d54af7e0c888503508f356fae4ff0e29
|
||||||
PYTHONPATH="." QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://gitlab.com/commaai/openpilot-lfs.git/gitlab-lfs/objects/c824f68646a3b94f117f01c70dc8316fb466e05fbd42ccdba440b8a8dc86914b
|
PYTHONPATH="." ASSERT_MIN_STEP_TIME=26 QCOM=1 taskset -c 4-7 python3 examples/openpilot/compile3.py https://gitlab.com/commaai/openpilot-lfs.git/gitlab-lfs/objects/c824f68646a3b94f117f01c70dc8316fb466e05fbd42ccdba440b8a8dc86914b
|
||||||
- name: benchmark MobileNetV2 on DSP
|
- name: benchmark MobileNetV2 on DSP
|
||||||
run: |
|
run: |
|
||||||
# generate quantized weights
|
# generate quantized weights
|
||||||
ln -s /data/home/tiny/tinygrad/extra/datasets/imagenet extra/datasets/imagenet
|
ln -s /data/home/tiny/tinygrad/extra/datasets/imagenet extra/datasets/imagenet
|
||||||
ln -s /data/home/tiny/tinygrad/testsig-*.so .
|
ln -s /data/home/tiny/tinygrad/testsig-*.so .
|
||||||
PYTHONPATH=. CC=clang-19 CPU=1 QUANT=1 CNT=0 python3 examples/test_onnx_imagenet.py https://github.com/xamcat/mobcat-samples/raw/refs/heads/master/onnx_runtime/InferencingSample/InferencingSample/mobilenetv2-7.onnx /tmp/model.quant.onnx
|
PYTHONPATH=. CC=clang-19 CPU=1 CPU_LLVM=0 QUANT=1 CNT=0 python3 examples/test_onnx_imagenet.py https://github.com/xamcat/mobcat-samples/raw/refs/heads/master/onnx_runtime/InferencingSample/InferencingSample/mobilenetv2-7.onnx /tmp/model.quant.onnx
|
||||||
# benchmark on DSP with NOOPT=1, the devectorizer has issues
|
# benchmark on DSP with NOOPT=1, the devectorizer has issues
|
||||||
PYTHONPATH=. CC=clang-19 DSP=1 DONT_REALIZE_EXPAND=1 NOOPT=1 CNT=2 DEBUG=2 python3 examples/test_onnx_imagenet.py /tmp/model.quant.onnx
|
PYTHONPATH=. CC=clang-19 DSP=1 DONT_REALIZE_EXPAND=1 NOOPT=1 CNT=2 DEBUG=2 python3 examples/test_onnx_imagenet.py /tmp/model.quant.onnx
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
@@ -643,7 +658,7 @@ jobs:
|
|||||||
testreddriverbenchmark:
|
testreddriverbenchmark:
|
||||||
name: AM Benchmark
|
name: AM Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinyboxrandom]
|
runs-on: [self-hosted, Linux, tinyboxrandom]
|
||||||
timeout-minutes: 15
|
timeout-minutes: 20
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -679,8 +694,8 @@ jobs:
|
|||||||
# Fails on 9070
|
# Fails on 9070
|
||||||
# - name: Test tensor cores
|
# - name: Test tensor cores
|
||||||
# run: |
|
# run: |
|
||||||
# AMD=1 AMD_LLVM=0 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded_amd TestLinearizer.test_tensor_cores_padded_uops
|
# AMD=1 AMD_LLVM=0 python3 test/test_linearizer.py test/opt/test_tensor_cores.py
|
||||||
# AMD=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded_amd TestLinearizer.test_tensor_cores_padded_uops
|
# AMD=1 AMD_LLVM=1 python3 test/test_linearizer.py test/opt/test_tensor_cores.py
|
||||||
# AMD=1 SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
|
# AMD=1 SHOULD_USE_TC=1 BFLOAT16=1 DEBUG=2 python3 extra/gemm/simple_matmul.py
|
||||||
- name: Run Tensor Core GEMM (AMD)
|
- name: Run Tensor Core GEMM (AMD)
|
||||||
run: AMD=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py | tee am_matmul_amd.txt
|
run: AMD=1 SHOULD_USE_TC=1 HALF=1 DEBUG=2 ATOL=2e-2 python3 extra/gemm/simple_matmul.py | tee am_matmul_amd.txt
|
||||||
@@ -688,8 +703,12 @@ jobs:
|
|||||||
run: DEBUG=2 AMD=1 python -m pytest -rA test/test_tiny.py
|
run: DEBUG=2 AMD=1 python -m pytest -rA test/test_tiny.py
|
||||||
- name: Test DISK copy time
|
- name: Test DISK copy time
|
||||||
run: AMD=1 TESTFILE=/raid/downloads/llama3-8b-sfr/model-00001-of-00004.safetensors python3 test/external/external_benchmark_disk_raw.py
|
run: AMD=1 TESTFILE=/raid/downloads/llama3-8b-sfr/model-00001-of-00004.safetensors python3 test/external/external_benchmark_disk_raw.py
|
||||||
|
- name: Test CPU copy time
|
||||||
|
run: |
|
||||||
|
AMD=1 GRAPH_ONE_KERNEL=1 PYTHONPATH=. NSZ=8192 python3 test/speed/external_test_copy_speed.py TestCopySpeed.testCopyDefaulttoCPUJit
|
||||||
|
AMD=1 GRAPH_ONE_KERNEL=1 PYTHONPATH=. NSZ=8192 python3 test/speed/external_test_copy_speed.py TestCopySpeed.testCopyCPUtoDefaultJit
|
||||||
- name: Run full CIFAR training w 1 GPU
|
- name: Run full CIFAR training w 1 GPU
|
||||||
run: time BENCHMARK_LOG=cifar AMD=1 DEFAULT_FLOAT=HALF LATEWINO=1 STEPS=1000 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee am_train_cifar_one_gpu.txt
|
run: time BENCHMARK_LOG=cifar AMD=1 DEFAULT_FLOAT=HALF STEPS=1000 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee am_train_cifar_one_gpu.txt
|
||||||
# TODO: enable
|
# TODO: enable
|
||||||
# - name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
# - name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
||||||
# run: BENCHMARK_LOG=resnet_10steps AMD=1 MNISTMOCK=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee am_train_resnet_one_gpu.txt
|
# run: BENCHMARK_LOG=resnet_10steps AMD=1 MNISTMOCK=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee am_train_resnet_one_gpu.txt
|
||||||
@@ -710,7 +729,7 @@ jobs:
|
|||||||
testgreendriverbenchmark:
|
testgreendriverbenchmark:
|
||||||
name: NV Benchmark
|
name: NV Benchmark
|
||||||
runs-on: [self-hosted, Linux, tinyboxrandom]
|
runs-on: [self-hosted, Linux, tinyboxrandom]
|
||||||
timeout-minutes: 15
|
timeout-minutes: 20
|
||||||
defaults:
|
defaults:
|
||||||
run:
|
run:
|
||||||
shell: bash -e -o pipefail {0}
|
shell: bash -e -o pipefail {0}
|
||||||
@@ -742,15 +761,19 @@ jobs:
|
|||||||
- name: Test driver start time
|
- name: Test driver start time
|
||||||
run: time DEBUG=3 NV=1 python3 test/test_tiny.py TestTiny.test_plus
|
run: time DEBUG=3 NV=1 python3 test/test_tiny.py TestTiny.test_plus
|
||||||
- name: Test tensor cores
|
- name: Test tensor cores
|
||||||
run: NV=1 ALLOW_TF32=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
run: NV=1 ALLOW_TF32=1 python3 test/opt/test_tensor_cores.py
|
||||||
- name: Test DISK copy time
|
- name: Test DISK copy time
|
||||||
run: NV=1 TESTFILE=/raid/downloads/llama3-8b-sfr/model-00001-of-00004.safetensors python3 test/external/external_benchmark_disk_raw.py
|
run: NV=1 TESTFILE=/raid/downloads/llama3-8b-sfr/model-00001-of-00004.safetensors python3 test/external/external_benchmark_disk_raw.py
|
||||||
|
- name: Test CPU copy time
|
||||||
|
run: |
|
||||||
|
NV=1 GRAPH_ONE_KERNEL=1 PYTHONPATH=. NSZ=8192 python3 test/speed/external_test_copy_speed.py TestCopySpeed.testCopyDefaulttoCPUJit
|
||||||
|
NV=1 GRAPH_ONE_KERNEL=1 PYTHONPATH=. NSZ=8192 python3 test/speed/external_test_copy_speed.py TestCopySpeed.testCopyCPUtoDefaultJit
|
||||||
- name: Test LLAMA-3
|
- name: Test LLAMA-3
|
||||||
run: BENCHMARK_LOG=llama3_beam NV=1 JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 examples/llama3.py --size 8B --benchmark --temperature 0 | tee nv_llama3_beam.txt
|
run: BENCHMARK_LOG=llama3_beam NV=1 JITBEAM=2 IGNORE_BEAM_CACHE=1 python3 examples/llama3.py --size 8B --benchmark --temperature 0 | tee nv_llama3_beam.txt
|
||||||
- name: Run full CIFAR training w 1 GPU
|
- name: Run full CIFAR training w 1 GPU
|
||||||
run: time BENCHMARK_LOG=cifar NV=1 DEFAULT_FLOAT=HALF LATEWINO=1 STEPS=1000 TARGET_EVAL_ACC_PCT=93.2 python3 examples/hlb_cifar10.py | tee nv_train_cifar_one_gpu.txt
|
run: time BENCHMARK_LOG=cifar NV=1 DEFAULT_FLOAT=HALF STEPS=1000 TARGET_EVAL_ACC_PCT=93.0 python3 examples/hlb_cifar10.py | tee nv_train_cifar_one_gpu.txt
|
||||||
- name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
#- name: Run 10 MLPerf ResNet50 training steps (1 gpu)
|
||||||
run: BENCHMARK_LOG=resnet_10steps NV=1 MNISTMOCK=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee nv_train_resnet_one_gpu.txt
|
# run: BENCHMARK_LOG=resnet_10steps NV=1 MNISTMOCK=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=256 GPUS=1 MODEL=resnet python3 examples/mlperf/model_train.py | tee nv_train_resnet_one_gpu.txt
|
||||||
- name: Run 10 MLPerf Bert training steps (1 gpu)
|
- name: Run 10 MLPerf Bert training steps (1 gpu)
|
||||||
# TODO: remove BERT_LAYERS once scheduler is fast
|
# TODO: remove BERT_LAYERS once scheduler is fast
|
||||||
run: BENCHMARK_LOG=bert_10steps NV=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=1 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py | tee nv_train_bert_one_gpu.txt
|
run: BENCHMARK_LOG=bert_10steps NV=1 CAPTURE_PROCESS_REPLAY=0 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=66 GPUS=1 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py | tee nv_train_bert_one_gpu.txt
|
||||||
|
|||||||
+254
-363
@@ -7,6 +7,7 @@ env:
|
|||||||
BUILD_CACHE_VERSION: '1'
|
BUILD_CACHE_VERSION: '1'
|
||||||
CAPTURE_PROCESS_REPLAY: 1
|
CAPTURE_PROCESS_REPLAY: 1
|
||||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
PYTHONPATH: ${{ github.workspace }}
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
@@ -29,12 +30,10 @@ jobs:
|
|||||||
key: llvm-speed
|
key: llvm-speed
|
||||||
deps: testing_minimal
|
deps: testing_minimal
|
||||||
llvm: 'true'
|
llvm: 'true'
|
||||||
- name: External Benchmark Schedule
|
|
||||||
run: PYTHONPATH="." python3 test/external/external_benchmark_schedule.py
|
|
||||||
- name: Speed Test
|
- name: Speed Test
|
||||||
run: LLVM=1 python3 test/speed/external_test_speed_v_torch.py
|
run: CPU=1 CPU_LLVM=1 python3 test/speed/external_test_speed_v_torch.py
|
||||||
- name: Speed Test (BEAM=2)
|
- name: Speed Test (BEAM=2)
|
||||||
run: BEAM=2 LLVM=1 python3 test/speed/external_test_speed_v_torch.py
|
run: BEAM=2 CPU=1 CPU_LLVM=1 python3 test/speed/external_test_speed_v_torch.py
|
||||||
|
|
||||||
docs:
|
docs:
|
||||||
name: Docs
|
name: Docs
|
||||||
@@ -47,7 +46,7 @@ jobs:
|
|||||||
uses: ./.github/actions/setup-tinygrad
|
uses: ./.github/actions/setup-tinygrad
|
||||||
with:
|
with:
|
||||||
deps: docs
|
deps: docs
|
||||||
pydeps: "capstone"
|
pydeps: "capstone torch"
|
||||||
- name: Build wheel and show size
|
- name: Build wheel and show size
|
||||||
run: |
|
run: |
|
||||||
pip install build
|
pip install build
|
||||||
@@ -71,98 +70,29 @@ jobs:
|
|||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
pip install $GITHUB_WORKSPACE
|
pip install $GITHUB_WORKSPACE
|
||||||
cp $GITHUB_WORKSPACE/examples/beautiful_mnist.py .
|
cp $GITHUB_WORKSPACE/examples/beautiful_mnist.py .
|
||||||
PYTHONPATH=$GITHUB_WORKSPACE BS=2 STEPS=10 python beautiful_mnist.py
|
BS=2 STEPS=10 python beautiful_mnist.py
|
||||||
- name: Test Docs Build
|
- name: Test Docs Build
|
||||||
run: python -m mkdocs build --strict
|
run: python -m mkdocs build --strict
|
||||||
- name: Test Docs
|
- name: Test Docs
|
||||||
run: |
|
run: |
|
||||||
python docs/abstractions2.py
|
python docs/abstractions2.py
|
||||||
python docs/abstractions3.py
|
python docs/abstractions3.py
|
||||||
|
- name: Test README
|
||||||
|
run: awk '/```python/{flag=1;next}/```/{flag=0}flag' README.md > README.py && python README.py
|
||||||
- name: Test Quickstart
|
- name: Test Quickstart
|
||||||
run: awk '/```python/{flag=1;next}/```/{flag=0}flag' docs/quickstart.md > quickstart.py && PYTHONPATH=. python quickstart.py
|
run: awk '/```python/{flag=1;next}/```/{flag=0}flag' docs/quickstart.md > quickstart.py && python quickstart.py
|
||||||
- name: Test DEBUG
|
- name: Test DEBUG
|
||||||
run: DEBUG=100 python3 -c "from tinygrad import Tensor; N = 1024; a, b = Tensor.rand(N, N), Tensor.rand(N, N); c = (a.reshape(N, 1, N) * b.T.reshape(1, N, N)).sum(axis=2); print((c.numpy() - (a.numpy() @ b.numpy())).mean())"
|
run: DEBUG=100 python3 -c "from tinygrad import Tensor; N = 1024; a, b = Tensor.rand(N, N), Tensor.rand(N, N); c = (a.reshape(N, 1, N) * b.T.reshape(1, N, N)).sum(axis=2); print((c.numpy() - (a.numpy() @ b.numpy())).mean())"
|
||||||
- name: Compile EfficientNet to C and test it
|
- name: Compile EfficientNet to C and test it
|
||||||
run: |
|
run: |
|
||||||
CPU=1 PYTHONPATH="." python examples/compile_efficientnet.py > recognize.c
|
CPU=1 CPU_LLVM=0 python examples/compile_efficientnet.py > recognize.c
|
||||||
clang -O2 recognize.c -lm -o recognize
|
clang -O2 recognize.c -lm -o recognize
|
||||||
cat test/models/efficientnet/Chicken.jpg | ./recognize | grep cock
|
cat test/models/efficientnet/Chicken.jpg | ./recognize | grep cock
|
||||||
|
|
||||||
autogen:
|
|
||||||
name: Autogen
|
|
||||||
runs-on: ubuntu-24.04
|
|
||||||
timeout-minutes: 15
|
|
||||||
steps:
|
|
||||||
- name: Checkout Code
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
- name: Setup Environment
|
|
||||||
uses: ./.github/actions/setup-tinygrad
|
|
||||||
with:
|
|
||||||
opencl: 'true'
|
|
||||||
amd: 'true'
|
|
||||||
cuda: 'true'
|
|
||||||
webgpu: 'true'
|
|
||||||
llvm: 'true'
|
|
||||||
- name: Install autogen support packages
|
|
||||||
run: sudo apt-get install -y --no-install-recommends llvm-14-dev libclang-14-dev
|
|
||||||
- name: Verify OpenCL autogen
|
|
||||||
run: |
|
|
||||||
cp tinygrad/runtime/autogen/opencl.py /tmp/opencl.py.bak
|
|
||||||
./autogen_stubs.sh opencl
|
|
||||||
diff /tmp/opencl.py.bak tinygrad/runtime/autogen/opencl.py
|
|
||||||
- name: Verify CUDA autogen
|
|
||||||
run: |
|
|
||||||
cp tinygrad/runtime/autogen/cuda.py /tmp/cuda.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/nv_gpu.py /tmp/nv_gpu.py.bak
|
|
||||||
./autogen_stubs.sh cuda
|
|
||||||
./autogen_stubs.sh nv
|
|
||||||
diff /tmp/cuda.py.bak tinygrad/runtime/autogen/cuda.py
|
|
||||||
diff /tmp/nv_gpu.py.bak tinygrad/runtime/autogen/nv_gpu.py
|
|
||||||
- name: Verify AMD autogen
|
|
||||||
run: |
|
|
||||||
cp tinygrad/runtime/autogen/hsa.py /tmp/hsa.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/kfd.py /tmp/kfd.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/comgr.py /tmp/comgr.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/amd_gpu.py /tmp/amd_gpu.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/sqtt.py /tmp/sqtt.py.bak
|
|
||||||
./autogen_stubs.sh hsa
|
|
||||||
./autogen_stubs.sh kfd
|
|
||||||
./autogen_stubs.sh comgr
|
|
||||||
./autogen_stubs.sh amd
|
|
||||||
./autogen_stubs.sh sqtt
|
|
||||||
diff /tmp/hsa.py.bak tinygrad/runtime/autogen/hsa.py
|
|
||||||
diff /tmp/kfd.py.bak tinygrad/runtime/autogen/kfd.py
|
|
||||||
diff /tmp/comgr.py.bak tinygrad/runtime/autogen/comgr.py
|
|
||||||
diff /tmp/amd_gpu.py.bak tinygrad/runtime/autogen/amd_gpu.py
|
|
||||||
diff /tmp/sqtt.py.bak tinygrad/runtime/autogen/sqtt.py
|
|
||||||
- name: Verify Linux autogen
|
|
||||||
run: |
|
|
||||||
cp tinygrad/runtime/autogen/libc.py /tmp/libc.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/io_uring.py /tmp/io_uring.py.bak
|
|
||||||
cp tinygrad/runtime/autogen/ib.py /tmp/ib.py.bak
|
|
||||||
./autogen_stubs.sh libc
|
|
||||||
./autogen_stubs.sh io_uring
|
|
||||||
./autogen_stubs.sh ib
|
|
||||||
diff /tmp/libc.py.bak tinygrad/runtime/autogen/libc.py
|
|
||||||
diff /tmp/io_uring.py.bak tinygrad/runtime/autogen/io_uring.py
|
|
||||||
diff /tmp/ib.py.bak tinygrad/runtime/autogen/ib.py
|
|
||||||
- name: Verify WebGPU autogen
|
|
||||||
run: |
|
|
||||||
cp tinygrad/runtime/autogen/webgpu.py /tmp/webgpu.py.bak
|
|
||||||
./autogen_stubs.sh webgpu
|
|
||||||
diff /tmp/webgpu.py.bak tinygrad/runtime/autogen/webgpu.py
|
|
||||||
- name: Verify LLVM autogen
|
|
||||||
run: |
|
|
||||||
cp tinygrad/runtime/autogen/llvm.py /tmp/llvm.py.bak
|
|
||||||
./autogen_stubs.sh llvm
|
|
||||||
diff /tmp/llvm.py.bak tinygrad/runtime/autogen/llvm.py
|
|
||||||
|
|
||||||
torchbackend:
|
torchbackend:
|
||||||
name: Torch Backend Tests
|
name: Torch Backend Tests
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -182,26 +112,24 @@ jobs:
|
|||||||
pip3 install --upgrade --force-reinstall ruff==0.11.0
|
pip3 install --upgrade --force-reinstall ruff==0.11.0
|
||||||
python3 -m ruff check extra/torch_backend/backend.py
|
python3 -m ruff check extra/torch_backend/backend.py
|
||||||
- name: Test one op
|
- name: Test one op
|
||||||
run: PYTHONPATH=. FORWARD_ONLY=1 TINY_BACKEND=1 python3 test/test_ops.py TestOps.test_add
|
run: FORWARD_ONLY=1 TINY_BACKEND=1 python3 test/test_ops.py TestOps.test_add
|
||||||
- name: Test ResNet-18
|
- name: Test ResNet-18
|
||||||
run: PYTHONPATH=. DEBUG=2 python3 extra/torch_backend/example.py
|
run: DEBUG=2 python3 extra/torch_backend/example.py
|
||||||
- name: My (custom) tests
|
- name: My (custom) tests
|
||||||
run: PYTHONPATH=. python3 extra/torch_backend/test.py
|
run: python3 extra/torch_backend/test.py
|
||||||
- name: Test one op in torch tests
|
- name: Test one op in torch tests
|
||||||
run: PYTHONPATH=. DEBUG=2 python3 extra/torch_backend/torch_tests.py TestTinyBackendPRIVATEUSE1.test_unary_log_tiny_float32
|
run: DEBUG=2 python3 extra/torch_backend/torch_tests.py TestTinyBackendPRIVATEUSE1.test_unary_log_tiny_float32
|
||||||
- name: Test Ops with TINY_BACKEND
|
- name: Test Ops with TINY_BACKEND
|
||||||
run: PYTHONPATH=. LLVM=1 LLVMOPT=0 TINY_BACKEND=1 python3 -m pytest -n auto test/test_ops.py --durations=20
|
run: CPU=1 CPU_LLVM=1 LLVMOPT=0 TINY_BACKEND=1 python3 -m pytest -n auto test/test_ops.py --durations=20
|
||||||
- name: Test in-place operations on views
|
- name: Test in-place operations on views
|
||||||
run: PYTHONPATH=. TORCH_DEBUG=1 python3 extra/torch_backend/test_inplace.py
|
run: TORCH_DEBUG=1 python3 extra/torch_backend/test_inplace.py
|
||||||
- name: Test multi-gpu
|
- name: Test multi-gpu
|
||||||
run: PYTHONPATH=. LLVM=1 GPUS=4 TORCH_DEBUG=1 python3 extra/torch_backend/test_multigpu.py
|
run: CPU=1 CPU_LLVM=1 GPUS=4 TORCH_DEBUG=1 python3 extra/torch_backend/test_multigpu.py
|
||||||
|
|
||||||
torchbackendmore:
|
torchbackendmore:
|
||||||
name: Torch Backend Tests More
|
name: Torch Backend Tests More
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -216,86 +144,14 @@ jobs:
|
|||||||
sudo apt update || true
|
sudo apt update || true
|
||||||
sudo apt install -y --no-install-recommends ninja-build
|
sudo apt install -y --no-install-recommends ninja-build
|
||||||
- name: Test beautiful_mnist in torch with TINY_BACKEND
|
- name: Test beautiful_mnist in torch with TINY_BACKEND
|
||||||
run: SPLIT_REDUCEOP=0 FUSE_ARANGE=1 PYTHONPATH=. LLVM=1 TARGET_EVAL_ACC_PCT=96.0 TINY_BACKEND=1 python3 examples/other_mnist/beautiful_mnist_torch.py
|
run: CPU=1 CPU_LLVM=1 TARGET_EVAL_ACC_PCT=96.0 TINY_BACKEND=1 python3 examples/other_mnist/beautiful_mnist_torch.py
|
||||||
- name: Test some torch tests (expect failure)
|
- name: Test some torch tests (expect failure)
|
||||||
run: PYTHONPATH=. python3 -m pytest extra/torch_backend/torch_tests.py -v --tb=no || true
|
run: python3 -m pytest extra/torch_backend/torch_tests.py -v --tb=no || true
|
||||||
|
|
||||||
tc:
|
|
||||||
name: Tensor Core tests
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
timeout-minutes: 10
|
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
|
||||||
- name: Checkout Code
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
- name: Setup Environment
|
|
||||||
uses: ./.github/actions/setup-tinygrad
|
|
||||||
with:
|
|
||||||
key: uops-minimal
|
|
||||||
deps: testing_minimal
|
|
||||||
- name: Test IMAGE=2 support
|
|
||||||
run: |
|
|
||||||
IMAGE=2 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm
|
|
||||||
IMAGE=2 PYTHON=1 python3 test/test_ops.py TestOps.test_simple_conv2d
|
|
||||||
- name: Test emulated METAL tensor cores
|
|
||||||
run: |
|
|
||||||
DEBUG=2 EMULATE_METAL=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_big_gemm
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_METAL=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_METAL=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
|
||||||
- name: Test emulated AMX tensor cores
|
|
||||||
run: PYTHONPATH=. DEBUG=2 AMX=1 EMULATE_AMX=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm
|
|
||||||
- name: Test emulated AMD tensor cores
|
|
||||||
run: |
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores_padded_amd TestLinearizer.test_tensor_cores_padded_uops
|
|
||||||
- name: Test emulated AMD MFMA tensor cores
|
|
||||||
run: |
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_MFMA=1 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_MFMA=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_MFMA=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
|
||||||
- name: Test emulated AMD RDNA4 tensor cores
|
|
||||||
run: |
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_RDNA4=1 FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_RDNA4=1 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_RDNA4=1 FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_RDNA4=1 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_RDNA4=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD_RDNA4=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
|
||||||
- name: Test emulated CUDA tensor cores
|
|
||||||
run: |
|
|
||||||
DEBUG=2 EMULATE_CUDA=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm_fp16
|
|
||||||
DEBUG=2 EMULATE_CUDA=1 ALLOW_TF32=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm
|
|
||||||
DEBUG=2 EMULATE_CUDA_SM75=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm_fp16
|
|
||||||
PYTHONPATH="." DEBUG=2 EMULATE_CUDA=1 ALLOW_TF32=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH="." DEBUG=2 EMULATE_CUDA=1 ALLOW_TF32=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_linearizer.py TestLinearizer.test_tensor_cores_padded TestLinearizer.test_tensor_cores_padded_uops
|
|
||||||
- name: Test emulated INTEL OpenCL tensor cores
|
|
||||||
run: DEBUG=2 EMULATE_INTEL=1 FORWARD_ONLY=1 PYTHON=1 HALF=1 N=64 python3 ./extra/gemm/simple_matmul.py
|
|
||||||
- name: Full test tensor cores
|
|
||||||
run: |
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_METAL=1 FORWARD_ONLY=1 PYTHON=1 python3 ./test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 FORWARD_ONLY=1 PYTHON=1 python3 ./test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_CUDA=1 ALLOW_TF32=1 FORWARD_ONLY=1 PYTHON=1 python3 ./test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_INTEL=1 FORWARD_ONLY=1 PYTHON=1 python3 ./test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
PYTHONPATH=. DEBUG=2 AMX=1 EMULATE_AMX=1 FORWARD_ONLY=1 PYTHON=1 python3 ./test/test_linearizer.py TestLinearizer.test_tensor_cores
|
|
||||||
- name: Test device flop counts
|
|
||||||
run: |
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_METAL=1 PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_AMD=1 PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_CUDA=1 PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
|
||||||
PYTHONPATH=. DEBUG=2 EMULATE_INTEL=1 PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
|
||||||
PYTHONPATH=. DEBUG=2 AMX=1 EMULATE_AMX=1 PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStats.test_simple_matmul
|
|
||||||
|
|
||||||
bepython:
|
bepython:
|
||||||
name: Python Backend
|
name: Python Backend
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
timeout-minutes: 10
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -305,15 +161,60 @@ jobs:
|
|||||||
key: be-minimal
|
key: be-minimal
|
||||||
deps: testing_minimal
|
deps: testing_minimal
|
||||||
- name: Test dtype with Python emulator
|
- name: Test dtype with Python emulator
|
||||||
run: DEBUG=1 PYTHONPATH=. PYTHON=1 python3 -m pytest -n=auto test/test_dtype.py test/test_dtype_alu.py
|
run: DEBUG=1 PYTHON=1 python3 -m pytest -n=auto test/test_dtype.py test/test_dtype_alu.py
|
||||||
- name: Test ops with Python emulator
|
- name: Test ops with Python emulator
|
||||||
run: DEBUG=2 PYTHON=1 python3 -m pytest -n=auto test/test_ops.py -k "not (test_split or test_simple_cumsum or test_cumsum or test_einsum or test_dot or test_dot_1d or test_big_gemm or test_broadcastdot or test_multidot or test_var_axis or test_std_axis or test_broadcast_full or test_broadcast_partial or test_simple_conv3d or test_dilated_conv_transpose2d or test_simple_conv_transpose3d or test_large_input_conv2d or test_max_pool2d or test_max_pool2d_simple or test_max_pool2d_bigger_stride or test_avg_pool2d or test_cat or test_scaled_product_attention or test_scaled_product_attention_causal or test_slice_fancy_indexing_dim_inject_none or test_slice_fancy_indexing_list_indices or test_slice_fancy_indexing_no_dim_collapse or test_slice_fancy_indexing_tuple_indices or test_slice_fancy_indexing_list_with_tensors or test_slice_fancy_indexing_dim_collapse_int or test_interpolate_bilinear or test_interpolate_bilinear_corners_aligned or test_scaled_dot_product_attention or test_cummax or test_simple_cummax or test_logcumsumexp or test_sort or test_cumprod)" --durations=20
|
run: DEBUG=2 SKIP_SLOW_TEST=1 PYTHON=1 python3 -m pytest -n=auto test/test_ops.py --durations=20
|
||||||
- name: Test uops with Python emulator
|
- name: Test uops with Python emulator
|
||||||
run: PYTHON=1 python3 -m pytest test/test_uops.py --durations=20
|
run: PYTHON=1 python3 -m pytest test/test_uops.py --durations=20
|
||||||
- name: Test symbolic with Python emulator
|
- name: Test symbolic with Python emulator
|
||||||
run: PYTHONPATH=. PYTHON=1 python3 test/test_symbolic_ops.py
|
run: PYTHON=1 python3 test/test_symbolic_ops.py
|
||||||
- name: test_renderer_failures with Python emulator
|
- name: test_renderer_failures with Python emulator
|
||||||
run: PYTHONPATH=. PYTHON=1 python3 -m pytest -rA test/test_renderer_failures.py::TestRendererFailures
|
run: PYTHON=1 python3 -m pytest -rA test/test_renderer_failures.py::TestRendererFailures
|
||||||
|
- name: Test IMAGE=2 support
|
||||||
|
run: |
|
||||||
|
IMAGE=2 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm
|
||||||
|
IMAGE=2 PYTHON=1 python3 test/test_ops.py TestOps.test_simple_conv2d
|
||||||
|
- name: Test emulated METAL tensor cores
|
||||||
|
run: |
|
||||||
|
DEBUG=2 EMULATE=METAL FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_big_gemm
|
||||||
|
DEBUG=2 EMULATE=METAL FORWARD_ONLY=1 PYTHON=1 python3 test/opt/test_tensor_cores.py
|
||||||
|
- name: Test emulated AMX tensor cores
|
||||||
|
run: DEBUG=2 AMX=1 EMULATE=AMX FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm
|
||||||
|
- name: Test emulated AMD tensor cores
|
||||||
|
run: |
|
||||||
|
DEBUG=2 EMULATE=AMD FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD FORWARD_ONLY=1 PYTHON=1 python3 test/opt/test_tensor_cores.py
|
||||||
|
- name: Test emulated AMD MFMA tensor cores
|
||||||
|
run: |
|
||||||
|
DEBUG=2 EMULATE=AMD_MFMA FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD_MFMA FORWARD_ONLY=1 PYTHON=1 python3 test/opt/test_tensor_cores.py
|
||||||
|
- name: Test emulated AMD RDNA4 tensor cores
|
||||||
|
run: |
|
||||||
|
DEBUG=2 EMULATE=AMD_RDNA4 FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD_RDNA4 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=0 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD_RDNA4 FORWARD_ONLY=1 PYTHON=1 N=16 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD_RDNA4 FORWARD_ONLY=1 PYTHON=1 N=64 HALF=1 ACC_HALF=1 ATOL=1e-3 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
DEBUG=2 EMULATE=AMD_RDNA4 FORWARD_ONLY=1 PYTHON=1 python3 test/opt/test_tensor_cores.py
|
||||||
|
- name: Test emulated CUDA tensor cores
|
||||||
|
run: |
|
||||||
|
DEBUG=2 EMULATE=CUDA FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm_fp16
|
||||||
|
DEBUG=2 EMULATE=CUDA ALLOW_TF32=1 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm
|
||||||
|
DEBUG=2 EMULATE=CUDA_SM75 FORWARD_ONLY=1 PYTHON=1 python3 test/test_ops.py TestOps.test_gemm_fp16
|
||||||
|
DEBUG=2 EMULATE=CUDA ALLOW_TF32=1 FORWARD_ONLY=1 PYTHON=1 python3 test/opt/test_tensor_cores.py
|
||||||
|
- name: Test emulated INTEL OpenCL tensor cores
|
||||||
|
run: DEBUG=2 EMULATE=INTEL FORWARD_ONLY=1 PYTHON=1 HALF=1 N=64 python3 ./extra/gemm/simple_matmul.py
|
||||||
|
- name: Test emulated AMX tensor cores
|
||||||
|
run: DEBUG=2 AMX=1 EMULATE=AMX FORWARD_ONLY=1 PYTHON=1 python3 test/opt/test_tensor_cores.py
|
||||||
|
- name: Test device flop counts
|
||||||
|
run: |
|
||||||
|
DEBUG=2 EMULATE=METAL PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
||||||
|
DEBUG=2 EMULATE=AMD PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
||||||
|
DEBUG=2 EMULATE=CUDA PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
||||||
|
DEBUG=2 EMULATE=INTEL PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStatsMatmulHalf
|
||||||
|
DEBUG=2 AMX=1 EMULATE=AMX PYTHON=1 python3 ./test/test_uops_stats.py TestUOpsStats.test_simple_matmul
|
||||||
|
|
||||||
linter:
|
linter:
|
||||||
name: Linters
|
name: Linters
|
||||||
@@ -343,6 +244,8 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
python -m mypy --strict-equality --lineprecision-report .
|
python -m mypy --strict-equality --lineprecision-report .
|
||||||
cat lineprecision.txt
|
cat lineprecision.txt
|
||||||
|
- name: Run TYPED=1
|
||||||
|
run: TYPED=1 python -c "import tinygrad"
|
||||||
|
|
||||||
unittest:
|
unittest:
|
||||||
name: Unit Tests
|
name: Unit Tests
|
||||||
@@ -356,32 +259,39 @@ jobs:
|
|||||||
uses: ./.github/actions/setup-tinygrad
|
uses: ./.github/actions/setup-tinygrad
|
||||||
with:
|
with:
|
||||||
key: unittest-12
|
key: unittest-12
|
||||||
pydeps: "pillow"
|
pydeps: "pillow numpy ftfy regex"
|
||||||
deps: testing_unit
|
deps: testing_unit
|
||||||
- name: Test README
|
- name: Check Device.DEFAULT
|
||||||
run: awk '/```python/{flag=1;next}/```/{flag=0}flag' README.md > README.py && PYTHONPATH=. python README.py
|
run: python -c "from tinygrad import Device; assert Device.DEFAULT == 'CPU', Device.DEFAULT"
|
||||||
- name: Run unit tests
|
- name: Run unit tests
|
||||||
run: PYTHONPATH="." python -m pytest -n=auto test/unit/ --durations=20
|
run: CPU=1 python -m pytest -n=auto test/unit/ --durations=20
|
||||||
|
- name: Check SPEC=1
|
||||||
|
run: SPEC=1 python3 test/test_tiny.py
|
||||||
- name: Run targetted tests on NULL backend
|
- name: Run targetted tests on NULL backend
|
||||||
run: PYTHONPATH="." NULL=1 python3 test/test_multitensor.py TestMultiTensor.test_data_parallel_resnet_train_step
|
run: NULL=1 python3 -m unittest test.test_multitensor.TestMultiTensor.test_data_parallel_resnet_train_step test/device/test_null.py
|
||||||
- name: Run SDXL on NULL backend
|
# TODO: too slow
|
||||||
run: MAX_BUFFER_SIZE=0 PYTHONPATH="." NULL=1 DEBUG=1 python3 examples/sdxl.py --seed 0 --noshow --timing --fakeweights
|
# - name: Run SDXL on NULL backend
|
||||||
|
# run: NULL=1 DEBUG=1 python3 examples/sdxl.py --seed 0 --noshow --timing --fakeweights
|
||||||
|
- name: Run Clip tests for SD MLPerf on NULL backend
|
||||||
|
run: NULL=1 python -m pytest -n=auto test/external/mlperf_stable_diffusion/external_test_models.py::TestOpenClip --durations=20
|
||||||
# TODO: support fake weights
|
# TODO: support fake weights
|
||||||
#- name: Run LLaMA 7B on 4 fake devices
|
#- name: Run LLaMA 7B on 4 fake devices
|
||||||
# run: NULL=1 python3 examples/llama.py --gen 1 --size 7B --shard 4 --prompt "Hello." --count 3 --temperature 0 --timing
|
# run: NULL=1 python3 examples/llama.py --gen 1 --size 7B --shard 4 --prompt "Hello." --count 3 --temperature 0 --timing
|
||||||
- name: Run GC tests
|
- name: Run GC tests
|
||||||
run: PYTHONPATH="." python test/external/external_uop_gc.py
|
run: python test/external/external_uop_gc.py
|
||||||
|
- name: External Benchmark Schedule
|
||||||
|
run: python3 test/external/external_benchmark_schedule.py
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
- name: Regen dataset on test_tiny
|
- name: Regen dataset on test_tiny
|
||||||
run: |
|
run: |
|
||||||
test/external/process_replay/reset.py
|
test/external/process_replay/reset.py
|
||||||
CAPTURE_PROCESS_REPLAY=1 python test/test_tiny.py TestTiny.test_plus
|
CAPTURE_PROCESS_REPLAY=1 python test/test_tiny.py TestTiny.test_plus
|
||||||
PYTHONPATH=. python extra/optimization/extract_dataset.py
|
python extra/optimization/extract_dataset.py
|
||||||
gzip -c /tmp/sops > extra/datasets/sops.gz
|
gzip -c /tmp/sops > extra/datasets/sops.gz
|
||||||
DEBUG=1 MIN_ASTS=1 PYTHONPATH=. python extra/optimization/get_action_space.py
|
#DEBUG=1 MIN_ASTS=1 python extra/optimization/get_action_space.py
|
||||||
- name: Repo line count < 17500 lines
|
- name: Repo line count < 18000 lines
|
||||||
run: MAX_LINE_COUNT=17500 python sz.py
|
run: MAX_LINE_COUNT=18000 python sz.py
|
||||||
|
|
||||||
fuzzing:
|
fuzzing:
|
||||||
name: Fuzzing
|
name: Fuzzing
|
||||||
@@ -400,18 +310,16 @@ jobs:
|
|||||||
- name: Fuzz Test fast idiv
|
- name: Fuzz Test fast idiv
|
||||||
run: python test/external/fuzz_fast_idiv.py
|
run: python test/external/fuzz_fast_idiv.py
|
||||||
- name: Fuzz Test shapetracker
|
- name: Fuzz Test shapetracker
|
||||||
run: |
|
run: CNT=50 python test/external/fuzz_shapetracker.py
|
||||||
PYTHONPATH="." python test/external/fuzz_shapetracker.py
|
- name: Fuzz Test shapetracker math
|
||||||
PYTHONPATH="." python test/external/fuzz_shapetracker_math.py
|
run: CNT=200 python test/external/fuzz_shapetracker_math.py
|
||||||
- name: Fuzz Test shape ops
|
- name: Fuzz Test shape ops
|
||||||
run: python test/external/fuzz_shape_ops.py
|
run: python test/external/fuzz_shape_ops.py
|
||||||
|
|
||||||
testgpuimage:
|
testopenclimage:
|
||||||
name: 'GPU IMAGE Tests'
|
name: CL IMAGE Tests
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 10
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -421,25 +329,17 @@ jobs:
|
|||||||
key: gpu-image
|
key: gpu-image
|
||||||
deps: testing_minimal
|
deps: testing_minimal
|
||||||
opencl: 'true'
|
opencl: 'true'
|
||||||
- name: Run Kernel Count Test
|
- name: Test CL IMAGE=2 ops + training
|
||||||
run: PYTHONPATH="." GPU=1 python -m pytest -n=auto test/external/external_test_opt.py
|
|
||||||
- name: Test WINO=1
|
|
||||||
run: GPU=1 DEBUG=2 WINO=1 python3 test/test_ops.py TestOps.test_simple_conv2d
|
|
||||||
- name: Test GPU IMAGE=2 ops + training
|
|
||||||
run: |
|
run: |
|
||||||
PYTHONPATH="." GPU=1 IMAGE=2 python -m pytest -n=auto test/test_ops.py --durations=20
|
CL=1 IMAGE=2 python -m pytest -n=auto test/test_ops.py --durations=20
|
||||||
PYTHONPATH="." GPU=1 IMAGE=2 python3 test/models/test_end2end.py TestEnd2End.test_linear_mnist
|
CL=1 IMAGE=2 python test/models/test_end2end.py TestEnd2End.test_linear_mnist
|
||||||
- name: Run fused optimizer tests
|
|
||||||
run: PYTHONPATH="." GPU=1 FUSE_OPTIM=1 python -m pytest -n=auto test/models/test_mnist.py
|
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
testgendataset:
|
testgpumisc:
|
||||||
name: 'GPU Generate Kernel Dataset'
|
name: CL Misc tests
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 10
|
timeout-minutes: 10
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -450,7 +350,11 @@ jobs:
|
|||||||
deps: testing_minimal
|
deps: testing_minimal
|
||||||
opencl: 'true'
|
opencl: 'true'
|
||||||
- name: Generate Dataset
|
- name: Generate Dataset
|
||||||
run: PYTHONPATH="." extra/optimization/generate_dataset.sh
|
run: CL=1 extra/optimization/generate_dataset.sh
|
||||||
|
- name: Run Kernel Count Test
|
||||||
|
run: CL=1 python -m pytest -n=auto test/external/external_test_opt.py
|
||||||
|
- name: Run fused optimizer tests
|
||||||
|
run: CL=1 FUSE_OPTIM=1 python -m pytest -n=auto test/models/test_mnist.py
|
||||||
- name: Upload artifact
|
- name: Upload artifact
|
||||||
uses: actions/upload-artifact@v4
|
uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
@@ -458,11 +362,9 @@ jobs:
|
|||||||
path: /tmp/sops.gz
|
path: /tmp/sops.gz
|
||||||
|
|
||||||
testopenpilot:
|
testopenpilot:
|
||||||
name: 'openpilot Compile Tests'
|
name: openpilot Compile Tests
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -475,26 +377,26 @@ jobs:
|
|||||||
llvm: 'true'
|
llvm: 'true'
|
||||||
- name: Test openpilot model kernel count and gate usage
|
- name: Test openpilot model kernel count and gate usage
|
||||||
run: |
|
run: |
|
||||||
PYTHONPATH="." ALLOWED_KERNEL_COUNT=208 ALLOWED_READ_IMAGE=2134 ALLOWED_GATED_READ_IMAGE=13 FLOAT16=0 GPU=1 IMAGE=2 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.4/selfdrive/modeld/models/supercombo.onnx
|
ALLOWED_KERNEL_COUNT=190 ALLOWED_READ_IMAGE=2081 ALLOWED_GATED_READ_IMAGE=28 FLOAT16=0 CL=1 IMAGE=2 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/v0.9.4/selfdrive/modeld/models/supercombo.onnx
|
||||||
- name: Test openpilot alt model correctness (float32)
|
- name: Test openpilot alt model correctness (float32)
|
||||||
run: PYTHONPATH="." FLOAT16=0 DEBUGCL=1 GPU=1 IMAGE=2 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/3799fe46b3a629e491d4b8498b8ae83e4c88c304/selfdrive/modeld/models/supercombo.onnx
|
run: FLOAT16=0 DEBUGCL=1 CL=1 IMAGE=2 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/3799fe46b3a629e491d4b8498b8ae83e4c88c304/selfdrive/modeld/models/supercombo.onnx
|
||||||
- name: Test openpilot fastvits model correctness (float32)
|
- name: Test openpilot fastvits model correctness (float32)
|
||||||
run: PYTHONPATH="." FLOAT16=0 DEBUGCL=1 GPU=1 IMAGE=2 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/9118973ed03c1ae1d40cf69a29507ec2cc78efd7/selfdrive/modeld/models/supercombo.onnx
|
run: FLOAT16=0 DEBUGCL=1 CL=1 IMAGE=2 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/9118973ed03c1ae1d40cf69a29507ec2cc78efd7/selfdrive/modeld/models/supercombo.onnx
|
||||||
# - name: Test openpilot simple_plan vision model correctness (float32)
|
# - name: Test openpilot simple_plan vision model correctness (float32)
|
||||||
# run: PYTHONPATH="." FLOAT16=0 DEBUGCL=1 GPU=1 IMAGE=2 python examples/openpilot/compile3.py https://gitlab.com/commaai/openpilot-lfs.git/gitlab-lfs/objects/35ff4f4577002f2685e50c8346addae33fe8da27a41dd4d6a0f14d1f4b1af81b
|
# run: FLOAT16=0 DEBUGCL=1 CL=1 IMAGE=2 python examples/openpilot/compile3.py https://gitlab.com/commaai/openpilot-lfs.git/gitlab-lfs/objects/35ff4f4577002f2685e50c8346addae33fe8da27a41dd4d6a0f14d1f4b1af81b
|
||||||
- name: Test openpilot LLVM compile
|
- name: Test openpilot LLVM compile
|
||||||
run: PYTHONPATH="." LLVM=1 LLVMOPT=1 JIT=2 BEAM=0 IMAGE=0 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/9118973ed03c1ae1d40cf69a29507ec2cc78efd7/selfdrive/modeld/models/supercombo.onnx
|
run: CPU=1 CPU_LLVM=1 LLVMOPT=1 JIT=2 BEAM=0 IMAGE=0 python examples/openpilot/compile3.py https://github.com/commaai/openpilot/raw/9118973ed03c1ae1d40cf69a29507ec2cc78efd7/selfdrive/modeld/models/supercombo.onnx
|
||||||
- name: Test openpilot compile4
|
- name: Test openpilot compile4
|
||||||
run: PYTHONPATH="." NOLOCALS=1 GPU=1 IMAGE=2 FLOAT16=1 DEBUG=2 python3 examples/openpilot/compile4.py
|
run: NOLOCALS=1 CL=1 IMAGE=2 FLOAT16=1 DEBUG=2 python3 examples/openpilot/compile4.py
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
|
# ****** ONNX Tests ******
|
||||||
|
|
||||||
testonnxcpu:
|
testonnxcpu:
|
||||||
name: 'ONNX (CPU) Tests'
|
name: ONNX (CPU) Tests
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
@@ -507,25 +409,22 @@ jobs:
|
|||||||
python-version: '3.11'
|
python-version: '3.11'
|
||||||
llvm: 'true'
|
llvm: 'true'
|
||||||
- name: Test ONNX (CPU)
|
- name: Test ONNX (CPU)
|
||||||
run: CPU=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
run: CPU=1 CPU_LLVM=0 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
||||||
- name: Test ONNX (LLVM)
|
- name: Test ONNX (LLVM)
|
||||||
run: LLVM=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
run: CPU=1 CPU_LLVM=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
||||||
- name: Test ONNX Runner (CPU)
|
- name: Test ONNX Runner (CPU)
|
||||||
run: CPU=1 PYTHONPATH=. python3 test/external/external_test_onnx_runner.py
|
run: CPU=1 CPU_LLVM=0 python3 test/external/external_test_onnx_runner.py
|
||||||
- name: Test Additional ONNX Ops (CPU)
|
- name: Test Additional ONNX Ops (CPU)
|
||||||
run: CPU=1 PYTHONPATH=. python3 test/external/external_test_onnx_ops.py
|
run: CPU=1 CPU_LLVM=0 python3 test/external/external_test_onnx_ops.py
|
||||||
- name: Test Quantize ONNX
|
- name: Test Quantize ONNX
|
||||||
run: CPU=1 PYTHONPATH=. python3 test/test_quantize_onnx.py
|
run: CPU=1 CPU_LLVM=0 python3 test/test_quantize_onnx.py
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
testopencl:
|
testopencl:
|
||||||
name: 'ONNX (GPU)+Optimization Tests'
|
name: ONNX (CL)+Optimization Tests
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -537,18 +436,22 @@ jobs:
|
|||||||
pydeps: "tensorflow==2.15.1 tensorflow_addons"
|
pydeps: "tensorflow==2.15.1 tensorflow_addons"
|
||||||
python-version: '3.11'
|
python-version: '3.11'
|
||||||
opencl: 'true'
|
opencl: 'true'
|
||||||
- name: Test ONNX (GPU)
|
- name: Test ONNX (CL)
|
||||||
run: GPU=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
run: CL=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
||||||
- name: Test Optimization Helpers
|
#- name: Test Optimization Helpers
|
||||||
run: PYTHONPATH="." DEBUG=1 python3 extra/optimization/test_helpers.py
|
# run: DEBUG=1 python3 extra/optimization/test_helpers.py
|
||||||
#- name: Test Action Space
|
#- name: Test Action Space
|
||||||
# run: PYTHONPATH="." DEBUG=1 GPU=1 python3 extra/optimization/get_action_space.py
|
# run: DEBUG=1 CL=1 python3 extra/optimization/get_action_space.py
|
||||||
- name: Test Beam Search
|
- name: Test Beam Search
|
||||||
run: PYTHONPATH="." GPU=1 IGNORE_BEAM_CACHE=1 python3 -m pytest extra/optimization/test_beam_search.py
|
run: CL=1 IGNORE_BEAM_CACHE=1 python3 -m pytest extra/optimization/test_beam_search.py
|
||||||
- name: Test MLPerf stuff
|
- name: Test MLPerf stuff
|
||||||
run: GPU=1 python -m pytest -n=auto test/external/external_test_optim.py test/external/external_test_losses.py test/external/external_test_metrics.py test/external/external_test_datasets.py --durations=20
|
run: CL=1 python -m pytest -n=auto test/external/external_test_optim.py test/external/external_test_losses.py test/external/external_test_metrics.py test/external/external_test_datasets.py --durations=20
|
||||||
|
- name: NULL=1 beautiful_mnist_multigpu
|
||||||
|
run: NULL=1 python examples/beautiful_mnist_multigpu.py
|
||||||
|
- name: Test Bert training
|
||||||
|
run: NULL=1 DEFAULT_FLOAT=HALF BENCHMARK=10 BS=24 GPUS=4 BERT_LAYERS=2 MODEL=bert python3 examples/mlperf/model_train.py
|
||||||
- name: Test llama 3 training
|
- name: Test llama 3 training
|
||||||
run: MAX_BUFFER_SIZE=0 PYTHONPATH="." DEV=NULL SAMPLES=300 BS=8 SEQLEN=512 GRADIENT_ACC_STEPS=8 FAKEDATA=1 DEFAULT_FLOAT=bfloat16 OPTIM_DTYPE=bfloat16 LLAMA3_SIZE=1B MODEL=llama3 python3 examples/mlperf/model_train.py
|
run: NULL=1 SAMPLES=300 BS=8 SEQLEN=512 GRADIENT_ACC_STEPS=8 FAKEDATA=1 DEFAULT_FLOAT=bfloat16 OPTIM_DTYPE=bfloat16 LLAMA3_SIZE=1B MODEL=llama3 python3 examples/mlperf/model_train.py
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
@@ -566,12 +469,12 @@ jobs:
|
|||||||
- name: Test 1B LLM
|
- name: Test 1B LLM
|
||||||
run: echo "What's a male chicken called? Answer with only one word." | MAX_BUFFER_SIZE=0 python3 -m tinygrad.apps.llm | grep -i rooster
|
run: echo "What's a male chicken called? Answer with only one word." | MAX_BUFFER_SIZE=0 python3 -m tinygrad.apps.llm | grep -i rooster
|
||||||
|
|
||||||
|
# ****** Models Tests ******
|
||||||
|
|
||||||
testmodels:
|
testmodels:
|
||||||
name: Models (llvm+cpu+gpu)
|
name: Models (llvm+cpu+gpu)
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -583,36 +486,38 @@ jobs:
|
|||||||
opencl: 'true'
|
opencl: 'true'
|
||||||
llvm: 'true'
|
llvm: 'true'
|
||||||
- name: Test models (llvm)
|
- name: Test models (llvm)
|
||||||
run: LLVM=1 python -m pytest -n=auto test/models --durations=20
|
run: CPU=1 CPU_LLVM=1 python -m pytest -n=auto test/models --durations=20
|
||||||
- name: Test models (gpu)
|
- name: Test models (opencl)
|
||||||
run: GPU=1 python -m pytest -n=auto test/models --durations=20
|
run: CL=1 python -m pytest -n=auto test/models --durations=20
|
||||||
- name: Test models (cpu)
|
- name: Test models (cpu)
|
||||||
run: CPU=1 python -m pytest -n=auto test/models --durations=20
|
run: CPU=1 CPU_LLVM=0 python -m pytest -n=auto test/models --durations=20
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
testrangeify:
|
testmetalmodels:
|
||||||
name: Linux (rangeify)
|
name: Models (metal)
|
||||||
runs-on: ubuntu-24.04
|
runs-on: macos-14
|
||||||
timeout-minutes: 15
|
timeout-minutes: 20
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
- name: Setup Environment
|
- name: Setup Environment
|
||||||
uses: ./.github/actions/setup-tinygrad
|
uses: ./.github/actions/setup-tinygrad
|
||||||
with:
|
with:
|
||||||
key: rangeify-minimal
|
key: metal
|
||||||
deps: testing_minimal
|
deps: testing
|
||||||
- name: Test CPU=1 RANGEIFY=1
|
python-version: '3.11'
|
||||||
# TODO: add more passing tests here
|
- name: Test models (Metal)
|
||||||
run: CPU=1 RANGEIFY=1 python3 -m pytest -n auto test/test_tiny.py test/test_rangeify.py test/test_ops.py --durations 20
|
run: METAL=1 python -m pytest -n=auto test/models --durations=20
|
||||||
|
- name: Test LLaMA compile speed
|
||||||
|
run: METAL=1 python test/external/external_test_speed_llama.py
|
||||||
|
|
||||||
|
# ****** Feature Tests ******
|
||||||
|
|
||||||
testdevectorize:
|
testdevectorize:
|
||||||
name: Linux (devectorize)
|
name: Linux (devectorize)
|
||||||
runs-on: ubuntu-24.04
|
runs-on: ubuntu-24.04
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -624,18 +529,16 @@ jobs:
|
|||||||
pydeps: "pillow"
|
pydeps: "pillow"
|
||||||
llvm: "true"
|
llvm: "true"
|
||||||
- name: Test LLVM=1 DEVECTORIZE=0
|
- name: Test LLVM=1 DEVECTORIZE=0
|
||||||
run: LLVM=1 DEVECTORIZE=0 python3 -m pytest -n auto test/test_tiny.py test/test_ops.py -k "not test_avg_pool3d_failure"
|
run: CPU=1 CPU_LLVM=1 DEVECTORIZE=0 python3 -m pytest -n auto test/test_tiny.py test/test_ops.py -k "not test_avg_pool3d_failure"
|
||||||
- name: Test LLVM=1 DEVECTORIZE=0 for model
|
- name: Test LLVM=1 DEVECTORIZE=0 for model
|
||||||
run: PYTHONPATH="." LLVM=1 DEVECTORIZE=0 python3 test/models/test_efficientnet.py
|
run: CPU=1 CPU_LLVM=1 DEVECTORIZE=0 python3 test/models/test_efficientnet.py
|
||||||
- name: Test CPU=1 DEVECTORIZE=0
|
- name: Test CPU=1 DEVECTORIZE=0
|
||||||
run: CPU=1 DEVECTORIZE=0 FUSE_ARANGE=0 python3 -m pytest -n auto test/test_tiny.py test/test_ops.py -k "not test_avg_pool3d_failure"
|
run: CPU=1 CPU_LLVM=0 DEVECTORIZE=0 python3 -m pytest -n auto test/test_tiny.py test/test_ops.py -k "not test_avg_pool3d_failure"
|
||||||
|
|
||||||
testdsp:
|
testdsp:
|
||||||
name: Linux (DSP)
|
name: Linux (DSP)
|
||||||
runs-on: ubuntu-24.04
|
runs-on: ubuntu-24.04
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -662,9 +565,9 @@ jobs:
|
|||||||
- name: Run test_tiny on DSP
|
- name: Run test_tiny on DSP
|
||||||
run: DEBUG=2 DSP=1 python test/test_tiny.py
|
run: DEBUG=2 DSP=1 python test/test_tiny.py
|
||||||
- name: Test transcendentals
|
- name: Test transcendentals
|
||||||
run: CC=clang-20 PYTHONPATH="." DEBUG=2 DSP=1 python test/test_transcendental.py TestTranscendentalVectorized
|
run: CC=clang-20 DEBUG=2 DSP=1 python test/test_transcendental.py TestTranscendentalVectorized
|
||||||
- name: Test quantize onnx
|
- name: Test quantize onnx
|
||||||
run: PYTHONPATH="." DEBUG=2 DSP=1 python3 test/test_quantize_onnx.py
|
run: DEBUG=2 DSP=1 python3 test/test_quantize_onnx.py
|
||||||
|
|
||||||
testwebgpu:
|
testwebgpu:
|
||||||
name: Linux (WebGPU)
|
name: Linux (WebGPU)
|
||||||
@@ -702,12 +605,10 @@ jobs:
|
|||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
env:
|
env:
|
||||||
IGNORE_OOB: 0
|
|
||||||
AMD: 1
|
AMD: 1
|
||||||
MOCKGPU: 1
|
MOCKGPU: 1
|
||||||
FORWARD_ONLY: 1
|
FORWARD_ONLY: 1
|
||||||
AMD_LLVM: ${{ matrix.backend == 'amdllvm' && '1' || matrix.backend != 'amdllvm' && '0' }}
|
AMD_LLVM: ${{ matrix.backend == 'amdllvm' && '1' || matrix.backend != 'amdllvm' && '0' }}
|
||||||
PYTHONPATH: ${{ github.workspace }}
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -733,7 +634,7 @@ jobs:
|
|||||||
run: TRANSCENDENTAL=2 python -m pytest -n=auto test/test_ops.py::TestOps::test_sin test/test_ops.py::TestOps::test_cos test/test_ops.py::TestOps::test_tan test/test_ops.py::TestOps::test_exp test/test_ops.py::TestOps::test_log --durations=20
|
run: TRANSCENDENTAL=2 python -m pytest -n=auto test/test_ops.py::TestOps::test_sin test/test_ops.py::TestOps::test_cos test/test_ops.py::TestOps::test_tan test/test_ops.py::TestOps::test_exp test/test_ops.py::TestOps::test_log --durations=20
|
||||||
- name: Run TestOps.test_add with SQTT
|
- name: Run TestOps.test_add with SQTT
|
||||||
run: |
|
run: |
|
||||||
PROFILE=1 SQTT=1 DEBUG=5 python3 test/test_ops.py TestOps.test_add
|
VIZ=1 SQTT=1 DEBUG=5 python3 test/test_ops.py TestOps.test_add
|
||||||
extra/sqtt/rgptool.py create "/tmp/profile.pkl.$USER" -o /tmp/gpu0.rgp
|
extra/sqtt/rgptool.py create "/tmp/profile.pkl.$USER" -o /tmp/gpu0.rgp
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
@@ -747,7 +648,9 @@ jobs:
|
|||||||
name: Linux (${{ matrix.backend }})
|
name: Linux (${{ matrix.backend }})
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
|
env:
|
||||||
|
MOCKGPU: 1
|
||||||
|
FORWARD_ONLY: 1
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -759,29 +662,26 @@ jobs:
|
|||||||
cuda: 'true'
|
cuda: 'true'
|
||||||
ocelot: 'true'
|
ocelot: 'true'
|
||||||
- name: Set env
|
- name: Set env
|
||||||
run: printf "${{ matrix.backend == 'PTX' && 'FORWARD_ONLY=1\nJIT=1\nOPT=2\nCUDA=1\nPTX=1\nMOCKGPU=1' || matrix.backend == 'nv' && 'NV=1\nMOCKGPU=1\nFORWARD_ONLY=1' }}" >> $GITHUB_ENV
|
run: printf "${{ matrix.backend == 'PTX' && 'CUDA=1\nCUDA_PTX=1' || matrix.backend == 'nv' && 'NV=1\nSKIP_SLOW_TEST=1' }}" >> $GITHUB_ENV
|
||||||
- name: Check Device.DEFAULT and print some source
|
- name: Check Device.DEFAULT and print some source
|
||||||
run: |
|
run: |
|
||||||
PYTHONPATH=${{ github.workspace }} python3 -c "from tinygrad import Device; assert Device.DEFAULT in ['CUDA','NV'], Device.DEFAULT"
|
python3 -c "from tinygrad import Device; assert Device.DEFAULT in ['CUDA','NV'], Device.DEFAULT"
|
||||||
DEBUG=5 PYTHONPATH=${{ github.workspace }} FORWARD_ONLY=1 python3 test/test_ops.py TestOps.test_add
|
DEBUG=5 FORWARD_ONLY=1 python3 test/test_ops.py TestOps.test_add
|
||||||
- name: Run pytest (cuda)
|
- name: Run pytest (cuda)
|
||||||
# skip multitensor because it's slow
|
# skip multitensor because it's slow
|
||||||
run: python -m pytest -n=auto test/ --ignore=test/models --ignore=test/unit --ignore test/test_gc.py --ignore test/test_multitensor.py --durations=20
|
run: python -m pytest -n=auto test/ --ignore=test/models --ignore=test/unit --ignore test/test_gc.py --ignore test/test_multitensor.py --durations=20
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
tests:
|
testcpuopencl:
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
backend: [llvm, cpu, gpu]
|
backend: [llvm, cpu, opencl]
|
||||||
|
|
||||||
name: Linux (${{ matrix.backend }})
|
name: Linux (${{ matrix.backend }})
|
||||||
runs-on: ubuntu-22.04
|
runs-on: ubuntu-22.04
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -790,65 +690,124 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
key: ${{ matrix.backend }}-minimal
|
key: ${{ matrix.backend }}-minimal
|
||||||
deps: testing_minimal
|
deps: testing_minimal
|
||||||
opencl: ${{ matrix.backend == 'gpu' && 'true' }}
|
opencl: ${{ matrix.backend == 'opencl' && 'true' }}
|
||||||
llvm: ${{ matrix.backend == 'llvm' && 'true' }}
|
llvm: ${{ matrix.backend == 'llvm' && 'true' }}
|
||||||
- name: Set env
|
- name: Set env
|
||||||
run: printf "${{ matrix.backend == 'llvm' && 'LLVM=1' || matrix.backend == 'cpu' && 'CPU=1' || matrix.backend == 'gpu' && 'GPU=1' }}" >> $GITHUB_ENV
|
run: printf "${{ matrix.backend == 'llvm' && 'CPU=1\nCPU_LLVM=1' || matrix.backend == 'cpu' && 'CPU=1\nCPU_LLVM=0\nCPU_COUNT=2' || matrix.backend == 'opencl' && 'CL=1' }}" >> $GITHUB_ENV
|
||||||
- name: Check Device.DEFAULT and print some source
|
- name: Check Device.DEFAULT and print some source
|
||||||
run: |
|
run: |
|
||||||
PYTHONPATH=${{ github.workspace }} python3 -c "from tinygrad import Device; assert Device.DEFAULT in ['LLVM','CPU','GPU'], Device.DEFAULT"
|
python3 -c "from tinygrad import Device; assert Device.DEFAULT in ['CPU','CL'], Device.DEFAULT"
|
||||||
DEBUG=5 PYTHONPATH=${{ github.workspace }} FORWARD_ONLY=1 python3 test/test_ops.py TestOps.test_add
|
DEBUG=5 FORWARD_ONLY=1 python3 test/test_ops.py TestOps.test_add
|
||||||
- name: Run pytest (not cuda)
|
- name: Run pytest (${{ matrix.backend }})
|
||||||
run: python -m pytest -n=auto test/ --ignore=test/models --ignore=test/unit --durations=20
|
run: python -m pytest -n=auto test/ --ignore=test/models --ignore=test/unit --durations=20
|
||||||
- name: Run TRANSCENDENTAL math
|
- name: Run TRANSCENDENTAL math
|
||||||
run: TRANSCENDENTAL=2 python -m pytest -n=auto test/test_ops.py::TestOps::test_sin test/test_ops.py::TestOps::test_cos test/test_ops.py::TestOps::test_tan test/test_ops.py::TestOps::test_exp test/test_ops.py::TestOps::test_log --durations=20
|
run: TRANSCENDENTAL=2 python -m pytest -n=auto test/test_ops.py::TestOps::test_sin test/test_ops.py::TestOps::test_cos test/test_ops.py::TestOps::test_tan test/test_ops.py::TestOps::test_exp test/test_ops.py::TestOps::test_log --durations=20
|
||||||
- name: Run process replay tests
|
- name: Run process replay tests
|
||||||
uses: ./.github/actions/process-replay
|
uses: ./.github/actions/process-replay
|
||||||
|
|
||||||
|
amdremote:
|
||||||
|
name: Linux (remote)
|
||||||
|
runs-on: ubuntu-22.04
|
||||||
|
timeout-minutes: 20
|
||||||
|
env:
|
||||||
|
REMOTE: 1
|
||||||
|
steps:
|
||||||
|
- name: Checkout Code
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
- name: Setup Environment
|
||||||
|
uses: ./.github/actions/setup-tinygrad
|
||||||
|
with:
|
||||||
|
key: linux-remote
|
||||||
|
deps: testing_minimal
|
||||||
|
amd: 'true'
|
||||||
|
llvm: 'true'
|
||||||
|
opencl: 'true'
|
||||||
|
- name: Start remote server
|
||||||
|
run: |
|
||||||
|
start_server() {
|
||||||
|
systemd-run --user \
|
||||||
|
--unit="$1" \
|
||||||
|
--setenv=REMOTEDEV="$2" \
|
||||||
|
--setenv=MOCKGPU=1 \
|
||||||
|
--setenv=PYTHONPATH=. \
|
||||||
|
--setenv=PORT="$3" \
|
||||||
|
--working-directory="$(pwd)" \
|
||||||
|
python tinygrad/runtime/ops_remote.py
|
||||||
|
}
|
||||||
|
|
||||||
|
start_server "remote-server-amd-1" "AMD" 6667
|
||||||
|
start_server "remote-server-amd-2" "AMD" 6668
|
||||||
|
start_server "remote-server-gpu" "CL" 7667
|
||||||
|
start_server "remote-server-cpu" "CPU" 8667
|
||||||
|
- name: Check Device.DEFAULT and print some source
|
||||||
|
env:
|
||||||
|
HOST: 127.0.0.1:6667*6,127.0.0.1:6668*6
|
||||||
|
run: |
|
||||||
|
python -c "from tinygrad import Device; assert Device.DEFAULT == 'REMOTE', Device.DEFAULT"
|
||||||
|
python -c "from tinygrad import Device; assert Device.default.properties.real_device == 'AMD', Device.default.properties.real_device"
|
||||||
|
DEBUG=4 python3 test/test_tiny.py TestTiny.test_plus
|
||||||
|
- name: Run REMOTE=1 Test (AMD)
|
||||||
|
env:
|
||||||
|
HOST: 127.0.0.1:6667*6,127.0.0.1:6668*6
|
||||||
|
run: |
|
||||||
|
python3 -m pytest test/test_tiny.py test/test_jit.py test/test_subbuffer.py test/test_graph.py test/test_multitensor.py test/test_remote.py test/test_tensor_variable.py --durations 20
|
||||||
|
- name: Run REMOTE=1 Test (CL)
|
||||||
|
env:
|
||||||
|
HOST: 127.0.0.1:7667*6
|
||||||
|
run: |
|
||||||
|
python3 -m pytest test/test_tiny.py test/test_image_dtype.py test/test_jit.py --durations 20
|
||||||
|
IMAGE=2 python3 -m pytest test/test_tiny.py test/test_image_dtype.py
|
||||||
|
- name: Run REMOTE=1 Test (CPU)
|
||||||
|
env:
|
||||||
|
HOST: 127.0.0.1:8667*6
|
||||||
|
run: |
|
||||||
|
python3 -m pytest test/test_tiny.py test/test_jit.py test/test_multitensor.py --durations 20
|
||||||
|
- name: Show remote server logs
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
journalctl --user -u remote-server-amd-1 --no-pager
|
||||||
|
journalctl --user -u remote-server-amd-2 --no-pager
|
||||||
|
journalctl --user -u remote-server-gpu --no-pager
|
||||||
|
journalctl --user -u remote-server-cpu --no-pager
|
||||||
|
|
||||||
# ****** OSX Tests ******
|
# ****** OSX Tests ******
|
||||||
|
|
||||||
testmetal2:
|
testmetal:
|
||||||
name: MacOS (unit)
|
name: MacOS (unit)
|
||||||
runs-on: macos-14
|
runs-on: macos-14
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
- name: Setup Environment
|
- name: Setup Environment
|
||||||
uses: ./.github/actions/setup-tinygrad
|
uses: ./.github/actions/setup-tinygrad
|
||||||
with:
|
with:
|
||||||
key: metal2
|
key: metal
|
||||||
deps: testing
|
deps: testing
|
||||||
python-version: '3.11'
|
python-version: '3.11'
|
||||||
amd: 'true'
|
amd: 'true'
|
||||||
cuda: 'true'
|
cuda: 'true'
|
||||||
ocelot: 'true'
|
ocelot: 'true'
|
||||||
llvm: 'true'
|
llvm: 'true'
|
||||||
- name: Run real world test
|
- name: Run unit tests
|
||||||
run: METAL=1 python -m pytest -n=auto test/models/test_real_world.py --durations=20
|
run: METAL=1 python -m pytest -n=auto test/unit/ --durations=20
|
||||||
- name: Test models (Metal)
|
|
||||||
run: METAL=1 python -m pytest -n=auto test/models -v --durations=20
|
|
||||||
- name: Run ONNX
|
- name: Run ONNX
|
||||||
run: METAL=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
run: METAL=1 python -m pytest -n=auto test/external/external_test_onnx_backend.py --durations=20
|
||||||
- name: Test tensor core ops (fake)
|
- name: Test tensor core ops (fake)
|
||||||
run: TC=2 METAL=1 DEBUG=3 python test/test_ops.py TestOps.test_gemm
|
run: METAL=1 DEBUG=3 TC=2 python test/test_ops.py TestOps.test_gemm
|
||||||
- name: Test tensor core ops (real)
|
- name: Test tensor core ops (real)
|
||||||
run: METAL=1 DEBUG=3 python test/test_ops.py TestOps.test_big_gemm
|
run: METAL=1 DEBUG=3 python test/test_ops.py TestOps.test_big_gemm
|
||||||
- name: Test LLaMA compile speed
|
|
||||||
run: PYTHONPATH="." METAL=1 python test/external/external_test_speed_llama.py
|
|
||||||
- name: Test Beam Search
|
- name: Test Beam Search
|
||||||
run: PYTHONPATH="." METAL=1 IGNORE_BEAM_CACHE=1 python3 -m pytest extra/optimization/test_beam_search.py
|
run: METAL=1 IGNORE_BEAM_CACHE=1 python3 -m pytest extra/optimization/test_beam_search.py
|
||||||
#- name: Fuzz Test linearizer
|
#- name: Fuzz Test linearizer
|
||||||
# run: PYTHONPATH="." METAL=1 DEPTH=4 FUZZ_N=50 FUZZ_MAX_SIZE=1000000 python test/external/fuzz_linearizer.py
|
# run: METAL=1 DEPTH=4 FUZZ_N=50 FUZZ_MAX_SIZE=1000000 python test/external/fuzz_linearizer.py
|
||||||
- name: Run TRANSCENDENTAL math
|
- name: Run TRANSCENDENTAL math
|
||||||
run: TRANSCENDENTAL=2 python -m pytest -n=auto test/test_ops.py::TestOps::test_sin test/test_ops.py::TestOps::test_cos test/test_ops.py::TestOps::test_tan test/test_ops.py::TestOps::test_exp test/test_ops.py::TestOps::test_log --durations=20
|
run: METAL=1 TRANSCENDENTAL=2 python -m pytest -n=auto test/test_ops.py::TestOps::test_sin test/test_ops.py::TestOps::test_cos test/test_ops.py::TestOps::test_tan test/test_ops.py::TestOps::test_exp test/test_ops.py::TestOps::test_log --durations=20
|
||||||
- name: Run pytest (amd)
|
- name: Run pytest (amd)
|
||||||
env:
|
env:
|
||||||
MOCKGPU: 1
|
MOCKGPU: 1
|
||||||
AMD: 1
|
AMD: 1
|
||||||
|
AMD_LLVM: 0
|
||||||
FORWARD_ONLY: 1
|
FORWARD_ONLY: 1
|
||||||
run: |
|
run: |
|
||||||
python3 -m pytest -n=auto test/device/test_hcq.py test/test_tiny.py --durations=20
|
python3 -m pytest -n=auto test/device/test_hcq.py test/test_tiny.py --durations=20
|
||||||
@@ -856,13 +815,14 @@ jobs:
|
|||||||
env:
|
env:
|
||||||
MOCKGPU: 1
|
MOCKGPU: 1
|
||||||
AMD: 1
|
AMD: 1
|
||||||
|
AMD_LLVM: 1
|
||||||
FORWARD_ONLY: 1
|
FORWARD_ONLY: 1
|
||||||
run: |
|
run: |
|
||||||
python -m pytest -n=auto test/device/test_hcq.py test/test_tiny.py test/device/test_amd_llvm.py --durations=20
|
python -m pytest -n=auto test/device/test_hcq.py test/test_tiny.py test/device/test_amd_llvm.py --durations=20
|
||||||
- name: Run pytest (ptx)
|
- name: Run pytest (ptx)
|
||||||
env:
|
env:
|
||||||
MOCKGPU: 1
|
MOCKGPU: 1
|
||||||
PTX: 1
|
NV_PTX: 1
|
||||||
NV: 1
|
NV: 1
|
||||||
FORWARD_ONLY: 1
|
FORWARD_ONLY: 1
|
||||||
run: |
|
run: |
|
||||||
@@ -905,7 +865,7 @@ jobs:
|
|||||||
# cp $GITHUB_WORKSPACE/test/web/test_viz.js .
|
# cp $GITHUB_WORKSPACE/test/web/test_viz.js .
|
||||||
# node test_viz.js
|
# node test_viz.js
|
||||||
- name: Test ONNX Runner (WEBGPU)
|
- name: Test ONNX Runner (WEBGPU)
|
||||||
run: WEBGPU=1 PYTHONPATH=. python3 test/external/external_test_onnx_runner.py
|
run: WEBGPU=1 python3 test/external/external_test_onnx_runner.py
|
||||||
|
|
||||||
osxremote:
|
osxremote:
|
||||||
name: MacOS (remote metal)
|
name: MacOS (remote metal)
|
||||||
@@ -931,72 +891,6 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
python3 -m pytest test/test_tiny.py test/test_jit.py test/test_subbuffer.py test/test_graph.py test/test_multitensor.py test/test_tensor_variable.py
|
python3 -m pytest test/test_tiny.py test/test_jit.py test/test_subbuffer.py test/test_graph.py test/test_multitensor.py test/test_tensor_variable.py
|
||||||
|
|
||||||
amdremote:
|
|
||||||
name: Linux (remote)
|
|
||||||
runs-on: ubuntu-22.04
|
|
||||||
timeout-minutes: 20
|
|
||||||
env:
|
|
||||||
REMOTE: 1
|
|
||||||
PYTHONPATH: ${{ github.workspace }}
|
|
||||||
steps:
|
|
||||||
- name: Checkout Code
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
- name: Setup Environment
|
|
||||||
uses: ./.github/actions/setup-tinygrad
|
|
||||||
with:
|
|
||||||
key: linux-remote
|
|
||||||
deps: testing_minimal
|
|
||||||
amd: 'true'
|
|
||||||
llvm: 'true'
|
|
||||||
opencl: 'true'
|
|
||||||
- name: Start remote server
|
|
||||||
run: |
|
|
||||||
start_server() {
|
|
||||||
systemd-run --user \
|
|
||||||
--unit="$1" \
|
|
||||||
--setenv=REMOTEDEV="$2" \
|
|
||||||
--setenv=MOCKGPU=1 \
|
|
||||||
--setenv=PYTHONPATH=. \
|
|
||||||
--setenv=PORT="$3" \
|
|
||||||
--working-directory="$(pwd)" \
|
|
||||||
python tinygrad/runtime/ops_remote.py
|
|
||||||
}
|
|
||||||
|
|
||||||
start_server "remote-server-amd-1" "AMD" 6667
|
|
||||||
start_server "remote-server-amd-2" "AMD" 6668
|
|
||||||
start_server "remote-server-gpu" "GPU" 7667
|
|
||||||
start_server "remote-server-cpu" "CPU" 8667
|
|
||||||
- name: Check Device.DEFAULT and print some source
|
|
||||||
env:
|
|
||||||
HOST: 127.0.0.1:6667*6,127.0.0.1:6668*6
|
|
||||||
run: |
|
|
||||||
python -c "from tinygrad import Device; assert Device.DEFAULT == 'REMOTE', Device.DEFAULT"
|
|
||||||
python -c "from tinygrad import Device; assert Device.default.properties.real_device == 'AMD', Device.default.properties.real_device"
|
|
||||||
DEBUG=4 python3 test/test_tiny.py TestTiny.test_plus
|
|
||||||
- name: Run REMOTE=1 Test (AMD)
|
|
||||||
env:
|
|
||||||
HOST: 127.0.0.1:6667*6,127.0.0.1:6668*6
|
|
||||||
run: |
|
|
||||||
python3 -m pytest test/test_tiny.py test/test_jit.py test/test_subbuffer.py test/test_graph.py test/test_multitensor.py test/test_remote.py test/test_tensor_variable.py --durations 20
|
|
||||||
- name: Run REMOTE=1 Test (GPU)
|
|
||||||
env:
|
|
||||||
HOST: 127.0.0.1:7667*6
|
|
||||||
run: |
|
|
||||||
python3 -m pytest test/test_tiny.py test/test_image_dtype.py test/test_jit.py --durations 20
|
|
||||||
IMAGE=2 python3 -m pytest test/test_tiny.py test/test_image_dtype.py
|
|
||||||
- name: Run REMOTE=1 Test (CPU)
|
|
||||||
env:
|
|
||||||
HOST: 127.0.0.1:8667*6
|
|
||||||
run: |
|
|
||||||
python3 -m pytest test/test_tiny.py test/test_jit.py test/test_multitensor.py --durations 20
|
|
||||||
- name: Show remote server logs
|
|
||||||
if: always()
|
|
||||||
run: |
|
|
||||||
journalctl --user -u remote-server-amd-1 --no-pager
|
|
||||||
journalctl --user -u remote-server-amd-2 --no-pager
|
|
||||||
journalctl --user -u remote-server-gpu --no-pager
|
|
||||||
journalctl --user -u remote-server-cpu --no-pager
|
|
||||||
|
|
||||||
osxtests:
|
osxtests:
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
@@ -1005,8 +899,6 @@ jobs:
|
|||||||
name: MacOS (${{ matrix.backend }})
|
name: MacOS (${{ matrix.backend }})
|
||||||
runs-on: macos-15
|
runs-on: macos-15
|
||||||
timeout-minutes: 20
|
timeout-minutes: 20
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -1018,10 +910,10 @@ jobs:
|
|||||||
pydeps: "capstone"
|
pydeps: "capstone"
|
||||||
llvm: ${{ matrix.backend == 'llvm' && 'true' }}
|
llvm: ${{ matrix.backend == 'llvm' && 'true' }}
|
||||||
- name: Set env
|
- name: Set env
|
||||||
run: printf "${{ matrix.backend == 'llvm' && 'LLVM=1' || matrix.backend == 'cpu' && 'CPU=1' || matrix.backend == 'metal' && 'METAL=1'}}" >> $GITHUB_ENV
|
run: printf "${{ matrix.backend == 'llvm' && 'CPU=1\nCPU_LLVM=1' || matrix.backend == 'cpu' && 'CPU=1\nCPU_LLVM=0\nCPU_COUNT=2' || matrix.backend == 'metal' && 'METAL=1'}}" >> $GITHUB_ENV
|
||||||
- name: Check Device.DEFAULT and print some source
|
- name: Check Device.DEFAULT and print some source
|
||||||
run: |
|
run: |
|
||||||
python -c "from tinygrad import Device; assert Device.DEFAULT == '${{ matrix.backend }}'.upper(), Device.DEFAULT"
|
python -c "from tinygrad import Device; assert Device.DEFAULT == {'LLVM':'CPU'}.get(x:='${{ matrix.backend }}'.upper(), x), Device.DEFAULT"
|
||||||
DEBUG=4 python3 test/test_tiny.py TestTiny.test_plus
|
DEBUG=4 python3 test/test_tiny.py TestTiny.test_plus
|
||||||
- name: Run pytest (${{ matrix.backend }})
|
- name: Run pytest (${{ matrix.backend }})
|
||||||
run: python3 -m pytest -n=auto test/ --ignore=test/models --ignore=test/unit --durations=20
|
run: python3 -m pytest -n=auto test/ --ignore=test/models --ignore=test/unit --durations=20
|
||||||
@@ -1042,8 +934,6 @@ jobs:
|
|||||||
name: Windows (${{ matrix.backend }})
|
name: Windows (${{ matrix.backend }})
|
||||||
runs-on: windows-latest
|
runs-on: windows-latest
|
||||||
timeout-minutes: 15
|
timeout-minutes: 15
|
||||||
env:
|
|
||||||
IGNORE_OOB: 0
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout Code
|
- name: Checkout Code
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
@@ -1055,12 +945,13 @@ jobs:
|
|||||||
pydeps: ${{ matrix.backend == 'webgpu' && 'dawn-python' || '' }}
|
pydeps: ${{ matrix.backend == 'webgpu' && 'dawn-python' || '' }}
|
||||||
- name: Set env
|
- name: Set env
|
||||||
shell: bash
|
shell: bash
|
||||||
run: printf "${{ matrix.backend == 'llvm' && 'LLVM=1' || matrix.backend == 'cpu' && 'CPU=1' || matrix.backend == 'webgpu' && 'WEBGPU=1'}}" >> $GITHUB_ENV
|
run: printf "${{ matrix.backend == 'llvm' && 'CPU=1\nCPU_LLVM=1' || matrix.backend == 'cpu' && 'CPU=1\nCPU_LLVM=0\nCPU_COUNT=2' || matrix.backend == 'webgpu' && 'WEBGPU=1'}}" >> $GITHUB_ENV
|
||||||
- name: Run unit tests
|
- name: Run unit tests
|
||||||
if: matrix.backend=='llvm'
|
if: matrix.backend=='llvm'
|
||||||
run: python -m pytest -n=auto test/unit/ --ignore=test/unit/test_disk_tensor.py --ignore=test/unit/test_elf.py --ignore=test/unit/test_tar.py
|
# test_newton_schulz hits RecursionError
|
||||||
|
run: python -m pytest -n=auto test/unit/ --ignore=test/unit/test_disk_tensor.py --ignore=test/unit/test_elf.py --ignore=test/unit/test_tar.py --ignore=test/unit/test_linalg.py --durations=20
|
||||||
- name: Run pytest (${{ matrix.backend }})
|
- name: Run pytest (${{ matrix.backend }})
|
||||||
shell: bash
|
shell: bash
|
||||||
run: |
|
run: |
|
||||||
python -c "from tinygrad import Device; assert Device.DEFAULT == '${{ matrix.backend }}'.upper(), Device.DEFAULT"
|
python -c "from tinygrad import Device; assert Device.DEFAULT == {'LLVM':'CPU'}.get(x:='${{ matrix.backend }}'.upper(), x), Device.DEFAULT"
|
||||||
python -m pytest -n=auto test/test_tiny.py test/test_ops.py --durations=20
|
python -m pytest -n=auto test/test_tiny.py test/test_ops.py --durations=20
|
||||||
|
|||||||
@@ -20,12 +20,6 @@ repos:
|
|||||||
language: system
|
language: system
|
||||||
always_run: true
|
always_run: true
|
||||||
pass_filenames: false
|
pass_filenames: false
|
||||||
- id: devicetests
|
|
||||||
name: select GPU tests
|
|
||||||
entry: env GPU=1 PYTHONPATH="." python3 -m pytest test/test_uops.py test/test_search.py
|
|
||||||
language: system
|
|
||||||
always_run: true
|
|
||||||
pass_filenames: false
|
|
||||||
- id: tests
|
- id: tests
|
||||||
name: subset of tests
|
name: subset of tests
|
||||||
entry: env PYTHONPATH="." python3 -m pytest -n=4 test/test_ops.py test/test_dtype.py test/test_schedule.py test/test_assign.py
|
entry: env PYTHONPATH="." python3 -m pytest -n=4 test/test_ops.py test/test_dtype.py test/test_schedule.py test/test_assign.py
|
||||||
|
|||||||
@@ -30,10 +30,6 @@ persistent=yes
|
|||||||
# Specify a configuration file.
|
# Specify a configuration file.
|
||||||
#rcfile=
|
#rcfile=
|
||||||
|
|
||||||
# When enabled, pylint would attempt to guess common misconfiguration and emit
|
|
||||||
# user-friendly hints instead of false-positive error messages
|
|
||||||
suggestion-mode=yes
|
|
||||||
|
|
||||||
# Allow loading of arbitrary C extensions. Extensions are imported into the
|
# Allow loading of arbitrary C extensions. Extensions are imported into the
|
||||||
# active Python interpreter and may run arbitrary code.
|
# active Python interpreter and may run arbitrary code.
|
||||||
unsafe-load-any-extension=no
|
unsafe-load-any-extension=no
|
||||||
@@ -54,11 +50,12 @@ confidence=
|
|||||||
# --enable=similarities". If you want to run only the classes checker, but have
|
# --enable=similarities". If you want to run only the classes checker, but have
|
||||||
# no Warning level messages displayed, use"--disable=all --enable=classes
|
# no Warning level messages displayed, use"--disable=all --enable=classes
|
||||||
# --disable=W"
|
# --disable=W"
|
||||||
disable=C,R,W0613,W0511,W0212,W0201,W0106,W0603,W0621,W0703,W1201,W1203,E1136,W1514,E1101,W0221,W0105,E0401,abstract-method
|
disable=C,R,W0613,W0511,W0212,W0201,W0106,W0603,W0621,W0703,W1201,W1203,E1136,W1514,E1101,W0221,W0105,E0401,abstract-method,W0707
|
||||||
# E1101 for function binding
|
# E1101 for function binding
|
||||||
# W0221 for Function class
|
# W0221 for Function class
|
||||||
# W0105 for comment strings
|
# W0105 for comment strings
|
||||||
# E0401 for missing imports
|
# E0401 for missing imports
|
||||||
|
# W0707 for not reraising
|
||||||
|
|
||||||
# Enable the message, report, category or checker with the given id(s). You can
|
# Enable the message, report, category or checker with the given id(s). You can
|
||||||
# either give multiple identifier separated by comma (,) or put this option
|
# either give multiple identifier separated by comma (,) or put this option
|
||||||
|
|||||||
@@ -79,9 +79,8 @@ See [examples/beautiful_mnist.py](examples/beautiful_mnist.py) for the full vers
|
|||||||
|
|
||||||
tinygrad already supports numerous accelerators, including:
|
tinygrad already supports numerous accelerators, including:
|
||||||
|
|
||||||
- [x] [GPU (OpenCL)](tinygrad/runtime/ops_gpu.py)
|
- [x] [OpenCL](tinygrad/runtime/ops_cl.py)
|
||||||
- [x] [CPU (C Code)](tinygrad/runtime/ops_cpu.py)
|
- [x] [CPU](tinygrad/runtime/ops_cpu.py)
|
||||||
- [x] [LLVM](tinygrad/runtime/ops_llvm.py)
|
|
||||||
- [x] [METAL](tinygrad/runtime/ops_metal.py)
|
- [x] [METAL](tinygrad/runtime/ops_metal.py)
|
||||||
- [x] [CUDA](tinygrad/runtime/ops_cuda.py)
|
- [x] [CUDA](tinygrad/runtime/ops_cuda.py)
|
||||||
- [x] [AMD](tinygrad/runtime/ops_amd.py)
|
- [x] [AMD](tinygrad/runtime/ops_amd.py)
|
||||||
|
|||||||
+20
-1
@@ -414,10 +414,29 @@ generate_sqtt() {
|
|||||||
clang2py -k cdefstum \
|
clang2py -k cdefstum \
|
||||||
extra/sqtt/sqtt.h \
|
extra/sqtt/sqtt.h \
|
||||||
-o $BASE/sqtt.py
|
-o $BASE/sqtt.py
|
||||||
|
|
||||||
fixup $BASE/sqtt.py
|
fixup $BASE/sqtt.py
|
||||||
sed -i "s\import ctypes\import ctypes, os\g" $BASE/sqtt.py
|
sed -i "s\import ctypes\import ctypes, os\g" $BASE/sqtt.py
|
||||||
python3 -c "import tinygrad.runtime.autogen.sqtt"
|
python3 -c "import tinygrad.runtime.autogen.sqtt"
|
||||||
|
|
||||||
|
ROCPROF_COMMIT_HASH=dd0485100971522cc4cd8ae136bdda431061a04d
|
||||||
|
ROCPROF_SRC=/tmp/rocprof-trace-decoder-$ROCPROF_COMMIT_HASH
|
||||||
|
if [ ! -d "$ROCPROF_SRC" ]; then
|
||||||
|
git clone https://github.com/ROCm/rocprof-trace-decoder $ROCPROF_SRC
|
||||||
|
pushd .
|
||||||
|
cd $ROCPROF_SRC
|
||||||
|
git reset --hard $ROCPROF_COMMIT_HASH
|
||||||
|
popd
|
||||||
|
fi
|
||||||
|
|
||||||
|
clang2py -k cdefstum \
|
||||||
|
$ROCPROF_SRC/include/rocprof_trace_decoder.h \
|
||||||
|
$ROCPROF_SRC/include/trace_decoder_instrument.h \
|
||||||
|
$ROCPROF_SRC/include/trace_decoder_types.h \
|
||||||
|
-o extra/sqtt/rocprof/rocprof.py
|
||||||
|
fixup extra/sqtt/rocprof/rocprof.py
|
||||||
|
sed -i '1s/^/# pylint: skip-file\n/' extra/sqtt/rocprof/rocprof.py
|
||||||
|
sed -i "s/import ctypes/import ctypes, ctypes.util/g" extra/sqtt/rocprof/rocprof.py
|
||||||
|
sed -i "s|FunctionFactoryStub()|ctypes.CDLL(ctypes.util.find_library('rocprof-trace-decoder'))|g" extra/sqtt/rocprof/rocprof.py
|
||||||
}
|
}
|
||||||
|
|
||||||
generate_webgpu() {
|
generate_webgpu() {
|
||||||
|
|||||||
@@ -42,7 +42,6 @@ import struct
|
|||||||
from tinygrad.dtype import dtypes
|
from tinygrad.dtype import dtypes
|
||||||
from tinygrad.device import Buffer, Device
|
from tinygrad.device import Buffer, Device
|
||||||
from tinygrad.uop.ops import UOp, Ops
|
from tinygrad.uop.ops import UOp, Ops
|
||||||
from tinygrad.shape.shapetracker import ShapeTracker
|
|
||||||
|
|
||||||
# allocate some buffers + load in values
|
# allocate some buffers + load in values
|
||||||
out = Buffer(DEVICE, 1, dtypes.int32).allocate()
|
out = Buffer(DEVICE, 1, dtypes.int32).allocate()
|
||||||
@@ -51,13 +50,14 @@ b = Buffer(DEVICE, 1, dtypes.int32).allocate().copyin(memoryview(bytearray(struc
|
|||||||
# NOTE: a._buf is the same as the return from cpu.allocator.alloc
|
# NOTE: a._buf is the same as the return from cpu.allocator.alloc
|
||||||
|
|
||||||
# describe the computation
|
# describe the computation
|
||||||
|
idx = UOp.const(dtypes.index, 0)
|
||||||
buf_1 = UOp(Ops.DEFINE_GLOBAL, dtypes.int32.ptr(), (), 1)
|
buf_1 = UOp(Ops.DEFINE_GLOBAL, dtypes.int32.ptr(), (), 1)
|
||||||
buf_2 = UOp(Ops.DEFINE_GLOBAL, dtypes.int32.ptr(), (), 2)
|
buf_2 = UOp(Ops.DEFINE_GLOBAL, dtypes.int32.ptr(), (), 2)
|
||||||
ld_1 = UOp(Ops.LOAD, dtypes.int32, (buf_1.view(ShapeTracker.from_shape((1,))),))
|
ld_1 = UOp(Ops.LOAD, dtypes.int32, (buf_1.index(idx),))
|
||||||
ld_2 = UOp(Ops.LOAD, dtypes.int32, (buf_2.view(ShapeTracker.from_shape((1,))),))
|
ld_2 = UOp(Ops.LOAD, dtypes.int32, (buf_2.index(idx),))
|
||||||
alu = ld_1 + ld_2
|
alu = ld_1 + ld_2
|
||||||
output_buf = UOp(Ops.DEFINE_GLOBAL, dtypes.int32.ptr(), (), 0)
|
output_buf = UOp(Ops.DEFINE_GLOBAL, dtypes.int32.ptr(), (), 0)
|
||||||
st_0 = UOp(Ops.STORE, dtypes.void, (output_buf.view(ShapeTracker.from_shape((1,))), alu))
|
st_0 = UOp(Ops.STORE, dtypes.void, (output_buf.index(idx), alu))
|
||||||
s = UOp(Ops.SINK, dtypes.void, (st_0,))
|
s = UOp(Ops.SINK, dtypes.void, (st_0,))
|
||||||
|
|
||||||
# convert the computation to a "linearized" format (print the format)
|
# convert the computation to a "linearized" format (print the format)
|
||||||
@@ -80,7 +80,7 @@ print("******** third, the UOp ***********")
|
|||||||
|
|
||||||
from tinygrad.engine.realize import run_schedule
|
from tinygrad.engine.realize import run_schedule
|
||||||
from tinygrad.engine.schedule import create_schedule_with_vars
|
from tinygrad.engine.schedule import create_schedule_with_vars
|
||||||
from tinygrad.schedule.kernelize import get_kernelize_map
|
from tinygrad.schedule.rangeify import get_rangeify_map
|
||||||
|
|
||||||
# allocate some values + load in values
|
# allocate some values + load in values
|
||||||
a = UOp.new_buffer(DEVICE, 1, dtypes.int32)
|
a = UOp.new_buffer(DEVICE, 1, dtypes.int32)
|
||||||
@@ -93,10 +93,10 @@ out = a + b
|
|||||||
s = UOp(Ops.SINK, dtypes.void, (out,))
|
s = UOp(Ops.SINK, dtypes.void, (out,))
|
||||||
|
|
||||||
# group the computation into kernels
|
# group the computation into kernels
|
||||||
becomes_map = get_kernelize_map(s)
|
becomes_map = get_rangeify_map(s)
|
||||||
|
|
||||||
# the compute maps to an assign
|
# the compute maps to an assign
|
||||||
assign = becomes_map[a+b]
|
assign = becomes_map[a+b].base
|
||||||
|
|
||||||
# the first source is the output buffer (data)
|
# the first source is the output buffer (data)
|
||||||
assert assign.src[0].op is Ops.BUFFER
|
assert assign.src[0].op is Ops.BUFFER
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ Directories are listed in order of how they are processed.
|
|||||||
|
|
||||||
Group UOps into kernels.
|
Group UOps into kernels.
|
||||||
|
|
||||||
::: tinygrad.schedule.kernelize.get_kernelize_map
|
::: tinygrad.schedule.rangeify.get_rangeify_map
|
||||||
options:
|
options:
|
||||||
members: false
|
members: false
|
||||||
show_labels: false
|
show_labels: false
|
||||||
@@ -22,12 +22,6 @@ Group UOps into kernels.
|
|||||||
|
|
||||||
Transforms the ast into an optimized ast. This is where BEAM search and heuristics live.
|
Transforms the ast into an optimized ast. This is where BEAM search and heuristics live.
|
||||||
|
|
||||||
::: tinygrad.codegen.opt.get_optimized_ast
|
|
||||||
options:
|
|
||||||
members: false
|
|
||||||
show_labels: false
|
|
||||||
show_source: false
|
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## tinygrad/codegen
|
## tinygrad/codegen
|
||||||
|
|||||||
+3
-6
@@ -3,7 +3,7 @@
|
|||||||
This is a list of environment variable that control the runtime behavior of tinygrad and its examples.
|
This is a list of environment variable that control the runtime behavior of tinygrad and its examples.
|
||||||
Most of these are self-explanatory, and are usually used to set an option at runtime.
|
Most of these are self-explanatory, and are usually used to set an option at runtime.
|
||||||
|
|
||||||
Example: `GPU=1 DEBUG=4 python3 -m pytest`
|
Example: `CL=1 DEBUG=4 python3 -m pytest`
|
||||||
|
|
||||||
However you can also decorate a function to set a value only inside that function.
|
However you can also decorate a function to set a value only inside that function.
|
||||||
|
|
||||||
@@ -31,19 +31,16 @@ These control the behavior of core tinygrad even when used as a library.
|
|||||||
Variable | Possible Value(s) | Description
|
Variable | Possible Value(s) | Description
|
||||||
---|---|---
|
---|---|---
|
||||||
DEBUG | [1-7] | enable debugging output (operations, timings, speed, generated code and more)
|
DEBUG | [1-7] | enable debugging output (operations, timings, speed, generated code and more)
|
||||||
GPU | [1] | enable the GPU (OpenCL) backend
|
CL | [1] | enable OpenCL backend
|
||||||
CUDA | [1] | enable CUDA backend
|
CUDA | [1] | enable CUDA backend
|
||||||
AMD | [1] | enable AMD backend
|
AMD | [1] | enable AMD backend
|
||||||
NV | [1] | enable NV backend
|
NV | [1] | enable NV backend
|
||||||
METAL | [1] | enable Metal backend (for Mac M1 and after)
|
METAL | [1] | enable Metal backend (for Mac M1 and after)
|
||||||
CPU | [1] | enable CPU (Clang) backend
|
CPU | [1] | enable CPU backend
|
||||||
LLVM | [1] | enable LLVM backend
|
|
||||||
BEAM | [#] | number of beams in kernel beam search
|
BEAM | [#] | number of beams in kernel beam search
|
||||||
DEFAULT_FLOAT | [HALF, ...]| specify the default float dtype (FLOAT32, HALF, BFLOAT16, FLOAT64, ...), default to FLOAT32
|
DEFAULT_FLOAT | [HALF, ...]| specify the default float dtype (FLOAT32, HALF, BFLOAT16, FLOAT64, ...), default to FLOAT32
|
||||||
IMAGE | [1-2] | enable 2d specific optimizations
|
IMAGE | [1-2] | enable 2d specific optimizations
|
||||||
FLOAT16 | [1] | use float16 for images instead of float32
|
FLOAT16 | [1] | use float16 for images instead of float32
|
||||||
PTX | [1] | enable the specialized [PTX](https://docs.nvidia.com/cuda/parallel-thread-execution/) assembler for Nvidia GPUs. If not set, defaults to generic CUDA codegen backend.
|
|
||||||
PROFILE | [1] | enable profiling. This feature is supported in NV, AMD, QCOM and METAL backends.
|
|
||||||
VISIBLE_DEVICES | [list[int]]| restricts the NV/AMD devices that are available. The format is a comma-separated list of identifiers (indexing starts with 0).
|
VISIBLE_DEVICES | [list[int]]| restricts the NV/AMD devices that are available. The format is a comma-separated list of identifiers (indexing starts with 0).
|
||||||
JIT | [0-2] | 0=disabled, 1=[jit enabled](quickstart.md#jit) (default), 2=jit enabled, but graphs are disabled
|
JIT | [0-2] | 0=disabled, 1=[jit enabled](quickstart.md#jit) (default), 2=jit enabled, but graphs are disabled
|
||||||
VIZ | [1] | 0=disabled, 1=[viz enabled](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/viz)
|
VIZ | [1] | 0=disabled, 1=[viz enabled](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/viz)
|
||||||
|
|||||||
+18
-11
@@ -2,17 +2,17 @@
|
|||||||
|
|
||||||
tinygrad supports various runtimes, enabling your code to scale across a wide range of devices. The default runtime can be automatically selected based on the available hardware, or you can force a specific runtime to be default using environment variables (e.g., `CPU=1`).
|
tinygrad supports various runtimes, enabling your code to scale across a wide range of devices. The default runtime can be automatically selected based on the available hardware, or you can force a specific runtime to be default using environment variables (e.g., `CPU=1`).
|
||||||
|
|
||||||
| Runtime | Description | Requirements |
|
| Runtime | Description | Compiler Options | Requirements |
|
||||||
|---------|-------------|--------------|
|
|---------|-------------|------------------|--------------|
|
||||||
| [NV](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_nv.py) | Provides acceleration for NVIDIA GPUs | Ampere/Ada series GPUs |
|
| [NV](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_nv.py) | Provides acceleration for NVIDIA GPUs | nvrtc (default)<br>PTX (`NV_PTX=1`) | Ampere/Ada/Blackwell series GPUs.<br>You can select an interface via `NV_IFACE=(NVK\|PCI)`. See [NV interfaces](#nv-interfaces) for details. |
|
||||||
| [AMD](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_amd.py) | Provides acceleration for AMD GPUs | RDNA2/RDNA3/RDNA4 series GPUs. You can select one of the interfaces for communication by setting `AMD_IFACE=(KFD|PCI)`. See [AMD interfaces](#amd-interfaces) for more details. |
|
| [AMD](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_amd.py) | Provides acceleration for AMD GPUs | LLVM (`AMD_LLVM=1`)<br>HIP/COMGR (`AMD_HIP=1`) | RDNA2 or newer GPUs.<br>You can select an interface via `AMD_IFACE=(KFD\|PCI\|USB)`. See [AMD interfaces](#amd-interfaces) for details. |
|
||||||
| [QCOM](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_qcom.py) | Provides acceleration for QCOM GPUs | 6xx series GPUs |
|
| [QCOM](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_qcom.py) | Provides acceleration for QCOM GPUs | - | 6xx series GPUs |
|
||||||
| [METAL](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_metal.py) | Utilizes Metal for acceleration on Apple devices | M1+ Macs; Metal 3.0+ for `bfloat` support |
|
| [METAL](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_metal.py) | Utilizes Metal for acceleration on Apple devices | - | M1+ Macs; Metal 3.0+ for `bfloat` support |
|
||||||
| [CUDA](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_cuda.py) | Utilizes CUDA for acceleration on NVIDIA GPUs | NVIDIA GPU with CUDA support |
|
| [CUDA](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_cuda.py) | Utilizes CUDA for acceleration on NVIDIA GPUs | nvrtc (default)<br> PTX (`CUDA_PTX=1`) | NVIDIA GPU with CUDA support |
|
||||||
| [GPU (OpenCL)](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_gpu.py) | Accelerates computations using OpenCL on GPUs | OpenCL 2.0 compatible device |
|
| [CL](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_cl.py) | Accelerates computations using OpenCL on GPUs | - | OpenCL 2.0 compatible device |
|
||||||
| [CPU (C Code)](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_cpu.py) | Runs on CPU using the clang compiler | `clang` compiler in system `PATH` |
|
| [CPU](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_cpu.py) | Runs on CPU using the clang or llvm compiler | Clang JIT (default)<br>LLVM IR (`CPU_LLVM=1`) | `clang` compiler in system `PATH` |
|
||||||
| [LLVM (LLVM IR)](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_llvm.py) | Runs on CPU using the LLVM compiler infrastructure | llvm libraries installed and findable |
|
| [WEBGPU](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_webgpu.py) | Runs on GPU using the Dawn WebGPU engine (used in Google Chrome) | - | Dawn library installed and discoverable. Binaries: [pydawn v0.3.0](https://github.com/wpmed92/pydawn/releases/tag/v0.3.0) |
|
||||||
| [WEBGPU](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/ops_webgpu.py) | Runs on GPU using the Dawn WebGPU engine (used in Google Chrome) | Dawn library installed and findable. Download binaries [here](https://github.com/wpmed92/pydawn/releases/tag/v0.3.0). |
|
|
||||||
|
|
||||||
## Interoperability
|
## Interoperability
|
||||||
|
|
||||||
@@ -70,5 +70,12 @@ AMD backend supports several interfaces for communicating with devices:
|
|||||||
|
|
||||||
* `KFD`: uses the amdgpu driver
|
* `KFD`: uses the amdgpu driver
|
||||||
* `PCI`: uses the [AM driver](developer/am.md)
|
* `PCI`: uses the [AM driver](developer/am.md)
|
||||||
|
* `USB`: USB3 interafce for asm24xx chips.
|
||||||
|
|
||||||
You can force an interface by setting `AMD_IFACE` to one of these values. In the case of `AMD_IFACE=PCI`, this may unbind your GPU from the amdgpu driver.
|
You can force an interface by setting `AMD_IFACE` to one of these values. In the case of `AMD_IFACE=PCI`, this may unbind your GPU from the amdgpu driver.
|
||||||
|
|
||||||
|
## NV Interfaces
|
||||||
|
NV backend supports several interfaces for communicating with devices:
|
||||||
|
|
||||||
|
* `NVK`: uses the nvidia driver
|
||||||
|
* `PCI`: uses the [NV driver](https://github.com/tinygrad/tinygrad/tree/master/tinygrad/runtime/support/nv/nvdev.py)
|
||||||
|
|||||||
@@ -78,6 +78,7 @@ Elementwise ops operate on a per element basis. They don't change the shape of t
|
|||||||
::: tinygrad.Tensor.minimum
|
::: tinygrad.Tensor.minimum
|
||||||
::: tinygrad.Tensor.where
|
::: tinygrad.Tensor.where
|
||||||
::: tinygrad.Tensor.copysign
|
::: tinygrad.Tensor.copysign
|
||||||
|
::: tinygrad.Tensor.logaddexp
|
||||||
|
|
||||||
## Casting Ops
|
## Casting Ops
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -6,7 +6,7 @@ If you don't have a tinybox and you want one, see [tinygrad.org](https://tinygra
|
|||||||
|
|
||||||
## Welcome
|
## Welcome
|
||||||
|
|
||||||
Welcome to your tinybox! The tinybox is the universal system purpose-built for all AI infrastructure and workloads, from training to inference. The red box includes six 7900XTX GPUs, and the green box includes six 4090 GPUs. Whether you bought a red one or a green one, we want you to love it.
|
Welcome to your tinybox! The tinybox is the universal system purpose-built for all AI infrastructure and workloads, from training to inference. The red box includes six 7900XTX GPUs, the green box includes six 4090 GPUs, and the green v2 box includes four 5090 GPUs. Whether you bought a red one or a green one, we want you to love it.
|
||||||
|
|
||||||
We don't have a stupid cloud service, you don't have to create a tiny account to set it up, and we aren't tracking how you use the box. We're just happy you bought one. This petaflop is your petaflop.
|
We don't have a stupid cloud service, you don't have to create a tiny account to set it up, and we aren't tracking how you use the box. We're just happy you bought one. This petaflop is your petaflop.
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,6 @@ import time
|
|||||||
start_tm = time.perf_counter()
|
start_tm = time.perf_counter()
|
||||||
import math
|
import math
|
||||||
from typing import Tuple, cast
|
from typing import Tuple, cast
|
||||||
import numpy as np
|
|
||||||
from tinygrad import Tensor, nn, GlobalCounters, TinyJit, dtypes, Device
|
from tinygrad import Tensor, nn, GlobalCounters, TinyJit, dtypes, Device
|
||||||
from tinygrad.helpers import partition, trange, getenv, Context
|
from tinygrad.helpers import partition, trange, getenv, Context
|
||||||
from extra.lr_scheduler import OneCycleLR
|
from extra.lr_scheduler import OneCycleLR
|
||||||
@@ -11,7 +10,7 @@ GPUS = [f'{Device.DEFAULT}:{i}' for i in range(getenv("GPUS", 1))]
|
|||||||
|
|
||||||
# override tinygrad defaults
|
# override tinygrad defaults
|
||||||
dtypes.default_float = dtypes.half
|
dtypes.default_float = dtypes.half
|
||||||
Context(FUSE_ARANGE=1, FUSE_OPTIM=1).__enter__()
|
Context(FUSE_OPTIM=1).__enter__()
|
||||||
|
|
||||||
# from https://github.com/tysam-code/hlb-CIFAR10/blob/main/main.py
|
# from https://github.com/tysam-code/hlb-CIFAR10/blob/main/main.py
|
||||||
batchsize = getenv("BS", 1024)
|
batchsize = getenv("BS", 1024)
|
||||||
@@ -150,13 +149,12 @@ if __name__ == "__main__":
|
|||||||
acc.append((out.argmax(-1) == Y).sum() / eval_batchsize)
|
acc.append((out.argmax(-1) == Y).sum() / eval_batchsize)
|
||||||
return Tensor.stack(*loss).mean() / (batchsize*loss_batchsize_scaler), Tensor.stack(*acc).mean()
|
return Tensor.stack(*loss).mean() / (batchsize*loss_batchsize_scaler), Tensor.stack(*acc).mean()
|
||||||
|
|
||||||
np.random.seed(1337)
|
Tensor.manual_seed(1337)
|
||||||
|
num_train_samples = X_train.shape[0]
|
||||||
|
|
||||||
for epoch in range(math.ceil(hyp['misc']['train_epochs'])):
|
for epoch in range(math.ceil(hyp['misc']['train_epochs'])):
|
||||||
# TODO: move to tinygrad
|
|
||||||
gst = time.perf_counter()
|
gst = time.perf_counter()
|
||||||
idxs = np.arange(X_train.shape[0])
|
tidxs = Tensor.randperm(num_train_samples, dtype='int')[:num_steps_per_epoch*batchsize].reshape(num_steps_per_epoch, batchsize)
|
||||||
np.random.shuffle(idxs)
|
|
||||||
tidxs = Tensor(idxs, dtype='int')[:num_steps_per_epoch*batchsize].reshape(num_steps_per_epoch, batchsize) # NOTE: long doesn't fold
|
|
||||||
train_loss:float = 0
|
train_loss:float = 0
|
||||||
for epoch_step in (t:=trange(num_steps_per_epoch)):
|
for epoch_step in (t:=trange(num_steps_per_epoch)):
|
||||||
st = time.perf_counter()
|
st = time.perf_counter()
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
import sys, time
|
import sys, time
|
||||||
from tinygrad import TinyJit, GlobalCounters, fetch, getenv
|
from tinygrad import TinyJit, GlobalCounters, fetch, getenv
|
||||||
from tinygrad.frontend.onnx import OnnxRunner
|
from tinygrad.nn.onnx import OnnxRunner
|
||||||
from extra.onnx_helpers import get_example_inputs, validate
|
from extra.onnx_helpers import get_example_inputs, validate
|
||||||
|
|
||||||
def load_onnx_model(onnx_file):
|
def load_onnx_model(onnx_file):
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ import numpy as np
|
|||||||
import subprocess
|
import subprocess
|
||||||
import tensorflow as tf
|
import tensorflow as tf
|
||||||
import tf2onnx
|
import tf2onnx
|
||||||
from tinygrad.frontend.onnx import OnnxRunner
|
from tinygrad.nn.onnx import OnnxRunner
|
||||||
from tinygrad.tensor import Tensor
|
from tinygrad.tensor import Tensor
|
||||||
from tinygrad.helpers import to_mv
|
from tinygrad.helpers import to_mv
|
||||||
from extra.export_model import export_model_clang, compile_net, jit_model
|
from extra.export_model import export_model_clang, compile_net, jit_model
|
||||||
|
|||||||
+13
-7
@@ -26,8 +26,8 @@ class Attention:
|
|||||||
start_pos = start_pos.val
|
start_pos = start_pos.val
|
||||||
|
|
||||||
if HALF: x = x.half()
|
if HALF: x = x.half()
|
||||||
xqkv = self.c_attn(x)
|
xqkv = self.c_attn(x).reshape(None, None, 3, self.n_heads, self.head_dim)
|
||||||
xq, xk, xv = [xqkv.shrink((None, None, (i*self.dim, (i+1)*self.dim))).reshape(None, None, self.n_heads, self.head_dim) for i in range(3)]
|
xq, xk, xv = [xqkv[:, :, i, :, :] for i in range(3)]
|
||||||
bsz, seqlen, _, _ = xq.shape
|
bsz, seqlen, _, _ = xq.shape
|
||||||
|
|
||||||
# create kv cache
|
# create kv cache
|
||||||
@@ -35,11 +35,11 @@ class Attention:
|
|||||||
self.cache_kv = Tensor.zeros(2, bsz, MAX_CONTEXT, self.n_heads, self.head_dim, dtype=x.dtype).contiguous().realize()
|
self.cache_kv = Tensor.zeros(2, bsz, MAX_CONTEXT, self.n_heads, self.head_dim, dtype=x.dtype).contiguous().realize()
|
||||||
|
|
||||||
# update the cache
|
# update the cache
|
||||||
self.cache_kv.shrink((None, None,(start_pos,start_pos+seqlen),None,None)).assign(Tensor.stack(xk, xv)).realize()
|
self.cache_kv[:, :, start_pos:start_pos+seqlen, :, :].assign(Tensor.stack(xk, xv)).realize()
|
||||||
|
|
||||||
if start_pos > 0:
|
if start_pos > 0:
|
||||||
keys = self.cache_kv[0].shrink((None, (0, start_pos+seqlen), None, None))
|
keys = self.cache_kv[0][:, :start_pos+seqlen, :, :]
|
||||||
values = self.cache_kv[1].shrink((None, (0, start_pos+seqlen), None, None))
|
values = self.cache_kv[1][:, :start_pos+seqlen, :, :]
|
||||||
else:
|
else:
|
||||||
keys = xk
|
keys = xk
|
||||||
values = xv
|
values = xv
|
||||||
@@ -64,7 +64,7 @@ class TransformerBlock:
|
|||||||
|
|
||||||
def __call__(self, x:Tensor, start_pos:Variable, mask:Optional[Tensor]):
|
def __call__(self, x:Tensor, start_pos:Variable, mask:Optional[Tensor]):
|
||||||
h = x + self.attn(self.ln_1(x), start_pos, mask).float()
|
h = x + self.attn(self.ln_1(x), start_pos, mask).float()
|
||||||
return (h + self.mlp(self.ln_2(h)))
|
return (h + self.mlp(self.ln_2(h))).contiguous()
|
||||||
|
|
||||||
class Transformer:
|
class Transformer:
|
||||||
def __init__(self, dim, n_heads, n_layers, norm_eps, vocab_size, max_seq_len=1024):
|
def __init__(self, dim, n_heads, n_layers, norm_eps, vocab_size, max_seq_len=1024):
|
||||||
@@ -181,6 +181,7 @@ class GPT2:
|
|||||||
self.tokenizer = tokenizer
|
self.tokenizer = tokenizer
|
||||||
|
|
||||||
def generate(self, prompt:str, max_length:int, temperature:float, timing:bool=False, batch_size:int=1):
|
def generate(self, prompt:str, max_length:int, temperature:float, timing:bool=False, batch_size:int=1):
|
||||||
|
step_times = []
|
||||||
prompt_tokens = self.tokenizer.encode(prompt, allowed_special={"<|endoftext|>"})
|
prompt_tokens = self.tokenizer.encode(prompt, allowed_special={"<|endoftext|>"})
|
||||||
toks = [prompt_tokens[:] for _ in range(batch_size)]
|
toks = [prompt_tokens[:] for _ in range(batch_size)]
|
||||||
start_pos = 0
|
start_pos = 0
|
||||||
@@ -188,7 +189,7 @@ class GPT2:
|
|||||||
GlobalCounters.reset()
|
GlobalCounters.reset()
|
||||||
if timing: print("")
|
if timing: print("")
|
||||||
st = GlobalCounters.time_sum_s
|
st = GlobalCounters.time_sum_s
|
||||||
with Timing("ran model in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on GPU" if DEBUG>=2 else "")+
|
with Timing("ran model in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on {Device.DEFAULT}" if DEBUG>=2 else "")+
|
||||||
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
||||||
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=timing):
|
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=timing):
|
||||||
with WallTimeEvent(BenchEvent.STEP):
|
with WallTimeEvent(BenchEvent.STEP):
|
||||||
@@ -197,8 +198,13 @@ class GPT2:
|
|||||||
else:
|
else:
|
||||||
tokens = Tensor([x[start_pos:] for x in toks])
|
tokens = Tensor([x[start_pos:] for x in toks])
|
||||||
tok = self.model(tokens, Variable("start_pos", 1 if start_pos else 0, MAX_CONTEXT-1).bind(start_pos), temperature).tolist()
|
tok = self.model(tokens, Variable("start_pos", 1 if start_pos else 0, MAX_CONTEXT-1).bind(start_pos), temperature).tolist()
|
||||||
|
step_times.append((GlobalCounters.time_sum_s-st)*1e3)
|
||||||
start_pos = len(toks[0])
|
start_pos = len(toks[0])
|
||||||
for i,t in enumerate(tok): toks[i].append(t)
|
for i,t in enumerate(tok): toks[i].append(t)
|
||||||
|
|
||||||
|
if (assert_time:=getenv("ASSERT_MIN_STEP_TIME")):
|
||||||
|
min_time = min(step_times)
|
||||||
|
assert min_time < assert_time, f"Speed regression, expected min step time of < {assert_time} ms but took: {min_time} ms"
|
||||||
return [self.tokenizer.decode(x) for x in toks]
|
return [self.tokenizer.decode(x) for x in toks]
|
||||||
|
|
||||||
# **** main code ****
|
# **** main code ****
|
||||||
|
|||||||
@@ -118,7 +118,7 @@ class SpeedyResNet:
|
|||||||
# hyper-parameters were exactly the same as the original repo
|
# hyper-parameters were exactly the same as the original repo
|
||||||
bias_scaler = 58
|
bias_scaler = 58
|
||||||
hyp = {
|
hyp = {
|
||||||
'seed' : 200,
|
'seed' : 201,
|
||||||
'opt': {
|
'opt': {
|
||||||
'bias_lr': 1.76 * bias_scaler/512,
|
'bias_lr': 1.76 * bias_scaler/512,
|
||||||
'non_bias_lr': 1.76 / 512,
|
'non_bias_lr': 1.76 / 512,
|
||||||
@@ -145,7 +145,6 @@ hyp = {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
@Context(FUSE_ARANGE=getenv("FUSE_ARANGE", 1))
|
|
||||||
def train_cifar():
|
def train_cifar():
|
||||||
|
|
||||||
def set_seed(seed):
|
def set_seed(seed):
|
||||||
@@ -229,7 +228,8 @@ def train_cifar():
|
|||||||
if getenv("RANDOM_CROP", 1):
|
if getenv("RANDOM_CROP", 1):
|
||||||
X = random_crop(X, crop_size=32)
|
X = random_crop(X, crop_size=32)
|
||||||
if getenv("RANDOM_FLIP", 1):
|
if getenv("RANDOM_FLIP", 1):
|
||||||
X = (Tensor.rand(X.shape[0],1,1,1) < 0.5).where(X.flip(-1), X) # flip LR
|
# NOTE: RANGEIFY=1 needs this contiguous or the X[perms] is very slow
|
||||||
|
X = (Tensor.rand(X.shape[0],1,1,1) < 0.5).where(X.flip(-1), X).contiguous() # flip LR
|
||||||
X, Y = X[perms], Y[perms]
|
X, Y = X[perms], Y[perms]
|
||||||
return X, Y, *cutmix(X, Y, perms, mask_size=hyp['net']['cutmix_size'])
|
return X, Y, *cutmix(X, Y, perms, mask_size=hyp['net']['cutmix_size'])
|
||||||
|
|
||||||
@@ -355,7 +355,7 @@ def train_cifar():
|
|||||||
|
|
||||||
# https://www.anandtech.com/show/16727/nvidia-announces-geforce-rtx-3080-ti-3070-ti-upgraded-cards-coming-in-june
|
# https://www.anandtech.com/show/16727/nvidia-announces-geforce-rtx-3080-ti-3070-ti-upgraded-cards-coming-in-june
|
||||||
# 136 TFLOPS is the theoretical max w float16 on 3080 Ti
|
# 136 TFLOPS is the theoretical max w float16 on 3080 Ti
|
||||||
|
step_times = []
|
||||||
model_ema: Optional[modelEMA] = None
|
model_ema: Optional[modelEMA] = None
|
||||||
projected_ema_decay_val = hyp['ema']['decay_base'] ** hyp['ema']['every_n_steps']
|
projected_ema_decay_val = hyp['ema']['decay_base'] ** hyp['ema']['every_n_steps']
|
||||||
i = 0
|
i = 0
|
||||||
@@ -413,12 +413,17 @@ def train_cifar():
|
|||||||
model_ema.update(model, Tensor([projected_ema_decay_val*(i/STEPS)**hyp['ema']['decay_pow']]))
|
model_ema.update(model, Tensor([projected_ema_decay_val*(i/STEPS)**hyp['ema']['decay_pow']]))
|
||||||
|
|
||||||
cl = time.monotonic()
|
cl = time.monotonic()
|
||||||
|
step_times.append((cl-st)*1000.0)
|
||||||
device_str = loss.device if isinstance(loss.device, str) else f"{loss.device[0]} * {len(loss.device)}"
|
device_str = loss.device if isinstance(loss.device, str) else f"{loss.device[0]} * {len(loss.device)}"
|
||||||
# 53 221.74 ms run, 2.22 ms python, 219.52 ms CL, 803.39 loss, 0.000807 LR, 4.66 GB used, 3042.49 GFLOPS, 674.65 GOPS
|
# 53 221.74 ms run, 2.22 ms python, 219.52 ms CL, 803.39 loss, 0.000807 LR, 4.66 GB used, 3042.49 GFLOPS, 674.65 GOPS
|
||||||
print(f"{i:3d} {(cl-st)*1000.0:7.2f} ms run, {(et-st)*1000.0:7.2f} ms python, {(cl-et)*1000.0:7.2f} ms {device_str}, {loss_cpu:7.2f} loss, {opt_non_bias.lr.numpy()[0]:.6f} LR, {GlobalCounters.mem_used/1e9:.2f} GB used, {GlobalCounters.global_ops*1e-9/(cl-st):9.2f} GFLOPS, {GlobalCounters.global_ops*1e-9:9.2f} GOPS")
|
print(f"{i:3d} {(cl-st)*1000.0:7.2f} ms run, {(et-st)*1000.0:7.2f} ms python, {(cl-et)*1000.0:7.2f} ms {device_str}, {loss_cpu:7.2f} loss, {opt_non_bias.lr.numpy()[0]:.6f} LR, {GlobalCounters.mem_used/1e9:.2f} GB used, {GlobalCounters.global_ops*1e-9/(cl-st):9.2f} GFLOPS, {GlobalCounters.global_ops*1e-9:9.2f} GOPS")
|
||||||
st = cl
|
st = cl
|
||||||
i += 1
|
i += 1
|
||||||
|
|
||||||
|
if (assert_time:=getenv("ASSERT_MIN_STEP_TIME")):
|
||||||
|
min_time = min(step_times)
|
||||||
|
assert min_time < assert_time, f"Speed regression, expected min step time of < {assert_time} ms but took: {min_time} ms"
|
||||||
|
|
||||||
# verify eval acc
|
# verify eval acc
|
||||||
if target := getenv("TARGET_EVAL_ACC_PCT", 0.0):
|
if target := getenv("TARGET_EVAL_ACC_PCT", 0.0):
|
||||||
if eval_acc_pct >= target:
|
if eval_acc_pct >= target:
|
||||||
|
|||||||
+1
-1
@@ -478,7 +478,7 @@ After you are done speaking, output [EOS]. You are not Chad.
|
|||||||
with Profiling(enabled=args.profile):
|
with Profiling(enabled=args.profile):
|
||||||
with Timing("total ", enabled=args.timing, on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
with Timing("total ", enabled=args.timing, on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
||||||
with WallTimeEvent(BenchEvent.STEP):
|
with WallTimeEvent(BenchEvent.STEP):
|
||||||
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on GPU" if DEBUG>=2 else "")+
|
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on {Device.DEFAULT}" if DEBUG>=2 else "")+
|
||||||
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
||||||
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=args.timing):
|
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=args.timing):
|
||||||
tok_tensor = llama.model(next_tok, start_pos, args.temperature)
|
tok_tensor = llama.model(next_tok, start_pos, args.temperature)
|
||||||
|
|||||||
+2
-2
@@ -441,7 +441,7 @@ if __name__ == "__main__":
|
|||||||
with Profiling(enabled=args.profile):
|
with Profiling(enabled=args.profile):
|
||||||
with Timing("total ", on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
with Timing("total ", on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
||||||
with WallTimeEvent(BenchEvent.STEP):
|
with WallTimeEvent(BenchEvent.STEP):
|
||||||
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on GPU" if DEBUG>=2 else "")+
|
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on {Device.DEFAULT}" if DEBUG>=2 else "")+
|
||||||
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
||||||
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None):
|
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None):
|
||||||
tok = model(Tensor([[last_tok]], device=device), start_pos, TEMPERATURE, TOP_K, TOP_P, ALPHA_F, ALPHA_P)
|
tok = model(Tensor([[last_tok]], device=device), start_pos, TEMPERATURE, TOP_K, TOP_P, ALPHA_F, ALPHA_P)
|
||||||
@@ -479,7 +479,7 @@ if __name__ == "__main__":
|
|||||||
st = GlobalCounters.time_sum_s
|
st = GlobalCounters.time_sum_s
|
||||||
with Profiling(enabled=args.profile):
|
with Profiling(enabled=args.profile):
|
||||||
with Timing("total ", enabled=args.timing, on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
with Timing("total ", enabled=args.timing, on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
||||||
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on GPU" if DEBUG>=2 else "")+
|
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on {Device.DEFAULT}" if DEBUG>=2 else "")+
|
||||||
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB"+
|
||||||
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=args.timing):
|
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=args.timing):
|
||||||
|
|
||||||
|
|||||||
+11
-3
@@ -279,9 +279,15 @@ def generate(model, tokenizer, prompt: str, n_tokens_to_gen: int = 10, temp: boo
|
|||||||
# Loading in the prompt tokens
|
# Loading in the prompt tokens
|
||||||
logits = model.forward(Tensor([tks]))[:, -1, :]
|
logits = model.forward(Tensor([tks]))[:, -1, :]
|
||||||
for _ in tqdm(range(n_tokens_to_gen), desc="Speed Gen"):
|
for _ in tqdm(range(n_tokens_to_gen), desc="Speed Gen"):
|
||||||
# TODO: topk
|
|
||||||
if sample:
|
if sample:
|
||||||
tok_Tens = (logits/temp).softmax().multinomial()
|
scaled_logits = logits / temp
|
||||||
|
if top_k is not None:
|
||||||
|
topk_values, topk_indices = scaled_logits.topk(top_k)
|
||||||
|
filtered_logits = Tensor.full_like(scaled_logits, -float("inf"))
|
||||||
|
filtered_logits = filtered_logits.scatter(dim=-1, index=topk_indices, src=topk_values)
|
||||||
|
tok_Tens = filtered_logits.softmax().multinomial()
|
||||||
|
else:
|
||||||
|
tok_Tens = scaled_logits.softmax().multinomial()
|
||||||
else:
|
else:
|
||||||
tok_Tens = logits.argmax(axis=-1).unsqueeze(0)
|
tok_Tens = logits.argmax(axis=-1).unsqueeze(0)
|
||||||
tok = tok_Tens.item()
|
tok = tok_Tens.item()
|
||||||
@@ -298,6 +304,7 @@ if __name__ == "__main__":
|
|||||||
parser.add_argument("--size", type=str, default="370m",
|
parser.add_argument("--size", type=str, default="370m",
|
||||||
help=f"Size of model to use [{', '.join([k for k in MODELS.keys()])}]")
|
help=f"Size of model to use [{', '.join([k for k in MODELS.keys()])}]")
|
||||||
parser.add_argument("--n_tokens", type=int, default=10, help="Number of tokens to generate")
|
parser.add_argument("--n_tokens", type=int, default=10, help="Number of tokens to generate")
|
||||||
|
parser.add_argument("--top_k", type=int, help="Limit sampling to the top k most likely tokens")
|
||||||
parser.add_argument("--sample", dest="sample", action="store_true", help="Sample flag")
|
parser.add_argument("--sample", dest="sample", action="store_true", help="Sample flag")
|
||||||
parser.add_argument("--temp", type=float, default=1.0, help="Sampling temp has to be <=1.0")
|
parser.add_argument("--temp", type=float, default=1.0, help="Sampling temp has to be <=1.0")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
@@ -308,8 +315,9 @@ if __name__ == "__main__":
|
|||||||
num_toks = args.n_tokens
|
num_toks = args.n_tokens
|
||||||
sample = args.sample
|
sample = args.sample
|
||||||
temp = args.temp
|
temp = args.temp
|
||||||
|
top_k = args.top_k
|
||||||
s = time.time()
|
s = time.time()
|
||||||
tinyoutput = generate(model, tokenizer, prompt, n_tokens_to_gen=num_toks, sample=sample, temp=temp)
|
tinyoutput = generate(model, tokenizer, prompt, n_tokens_to_gen=num_toks, sample=sample, temp=temp, top_k=top_k)
|
||||||
print(tinyoutput)
|
print(tinyoutput)
|
||||||
print('TIME: ', time.time() - s)
|
print('TIME: ', time.time() - s)
|
||||||
TORCHOUTPUT = "Why is gravity \nso important?\nBecause it's the only"
|
TORCHOUTPUT = "Why is gravity \nso important?\nBecause it's the only"
|
||||||
|
|||||||
@@ -511,6 +511,33 @@ def batch_load_retinanet(dataset, val:bool, base_dir:Path, batch_size:int=32, sh
|
|||||||
# happens with BENCHMARK set
|
# happens with BENCHMARK set
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
# stable diffusion callbacks to match mlperf ref; declared here because they're pickled
|
||||||
|
def filter_dataset(sample:dict): return {k:v for k,v in sample.items() if k in {'npy', 'txt'}}
|
||||||
|
def collate(batch:list[dict]):
|
||||||
|
ret = {"npy": [], "txt": [], "__key__": []}
|
||||||
|
for sample in batch:
|
||||||
|
for k,v in sample.items():
|
||||||
|
ret[k].append(v)
|
||||||
|
return ret
|
||||||
|
def collate_fn(batch): return batch
|
||||||
|
|
||||||
|
# Reference (code): https://github.com/mlcommons/training/blob/2f4a93fb4888180755a8ef55f4b977ef8f60a89e/stable_diffusion/ldm/data/webdatasets.py, Line 55
|
||||||
|
# Reference (params): https://github.com/mlcommons/training/blob/ab4ae1ca718d7fe62c369710a316dff18768d04b/stable_diffusion/configs/train_01x08x08.yaml, Line 107
|
||||||
|
def batch_load_train_stable_diffusion(urls:str, BS:int):
|
||||||
|
import webdataset
|
||||||
|
dataset = webdataset.WebDataset(urls=urls, resampled=True, cache_size=-1, cache_dir=None)
|
||||||
|
dataset = dataset.shuffle(size=1000)
|
||||||
|
dataset = dataset.decode()
|
||||||
|
dataset = dataset.map(filter_dataset)
|
||||||
|
dataset = dataset.batched(BS, partial=False, collation_fn=collate)
|
||||||
|
dataset = webdataset.WebLoader(dataset, batch_size=None, shuffle=False, num_workers=1, persistent_workers=True, collate_fn=collate_fn)
|
||||||
|
|
||||||
|
for x in dataset:
|
||||||
|
assert isinstance(x, dict) and all(isinstance(k, str) for k in x.keys()) and all(isinstance(v, list) for v in x.values())
|
||||||
|
assert all(isinstance(moment_mean_logvar, np.ndarray) and moment_mean_logvar.shape==(1,8,64,64) for moment_mean_logvar in x["npy"])
|
||||||
|
assert all(isinstance(caption, str) for caption in x["txt"])
|
||||||
|
yield x
|
||||||
|
|
||||||
# llama3
|
# llama3
|
||||||
|
|
||||||
class BinIdxDataset:
|
class BinIdxDataset:
|
||||||
@@ -758,6 +785,27 @@ def batch_load_llama3(bs:int, samples:int, seqlen:int, base_dir:Path, seed:int=0
|
|||||||
batch.append(tokens)
|
batch.append(tokens)
|
||||||
yield Tensor.stack(batch, dim=0)
|
yield Tensor.stack(batch, dim=0)
|
||||||
|
|
||||||
|
def batch_load_llama3_small(bs:int, samples:int, seqlen:int, base_dir:Path, seed:int=0, val:bool=True):
|
||||||
|
if val:
|
||||||
|
dataset = BlendedGPTDataset([
|
||||||
|
base_dir / "c4-validation-91205-samples.en_text_document",
|
||||||
|
], [
|
||||||
|
1.0
|
||||||
|
], samples, seqlen, seed, False)
|
||||||
|
else:
|
||||||
|
dataset = BlendedGPTDataset([
|
||||||
|
base_dir / "c4-train.en_6_text_document",
|
||||||
|
], [
|
||||||
|
1.0
|
||||||
|
], samples, seqlen, seed, True)
|
||||||
|
|
||||||
|
for b in range(math.ceil(samples / bs)):
|
||||||
|
batch = []
|
||||||
|
for i in range(bs):
|
||||||
|
tokens = dataset.get(b * bs + i)
|
||||||
|
batch.append(tokens)
|
||||||
|
yield Tensor.stack(batch, dim=0)
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
def load_unet3d(val):
|
def load_unet3d(val):
|
||||||
assert not val, "validation set is not supported due to different sizes on inputs"
|
assert not val, "validation set is not supported due to different sizes on inputs"
|
||||||
|
|||||||
@@ -2,7 +2,9 @@ import math
|
|||||||
from typing import Union
|
from typing import Union
|
||||||
|
|
||||||
from tinygrad import Tensor, nn, dtypes
|
from tinygrad import Tensor, nn, dtypes
|
||||||
from tinygrad.helpers import prod, argfix
|
from tinygrad.helpers import prod, argfix, Context
|
||||||
|
from tinygrad.nn.state import get_parameters
|
||||||
|
from extra.models.unet import UNetModel
|
||||||
|
|
||||||
# rejection sampling truncated randn
|
# rejection sampling truncated randn
|
||||||
def rand_truncn(*shape, dtype=None, truncstds=2, **kwargs) -> Tensor:
|
def rand_truncn(*shape, dtype=None, truncstds=2, **kwargs) -> Tensor:
|
||||||
@@ -17,6 +19,10 @@ def he_normal(*shape, a: float = 0.00, **kwargs) -> Tensor:
|
|||||||
std = math.sqrt(2.0 / (1 + a ** 2)) / math.sqrt(prod(argfix(*shape)[1:])) / 0.87962566103423978
|
std = math.sqrt(2.0 / (1 + a ** 2)) / math.sqrt(prod(argfix(*shape)[1:])) / 0.87962566103423978
|
||||||
return std * rand_truncn(*shape, **kwargs)
|
return std * rand_truncn(*shape, **kwargs)
|
||||||
|
|
||||||
|
# Stable Diffusion v2 training uses default torch gelu, which doesn't use tanh approximation
|
||||||
|
def gelu_erf(x:Tensor) -> Tensor:
|
||||||
|
return 0.5 * x * (1.0 + (x / 1.4142135623730951).erf())
|
||||||
|
|
||||||
class Conv2dHeNormal(nn.Conv2d):
|
class Conv2dHeNormal(nn.Conv2d):
|
||||||
def __init__(self, in_channels, out_channels, kernel_size, stride=1, padding=0, dilation=1, groups=1, bias=True):
|
def __init__(self, in_channels, out_channels, kernel_size, stride=1, padding=0, dilation=1, groups=1, bias=True):
|
||||||
super().__init__(in_channels, out_channels, kernel_size, stride=stride, padding=padding, dilation=dilation, groups=groups, bias=bias)
|
super().__init__(in_channels, out_channels, kernel_size, stride=stride, padding=padding, dilation=dilation, groups=groups, bias=bias)
|
||||||
@@ -127,3 +133,59 @@ class Conv2dRetinaNet(nn.Conv2d):
|
|||||||
def __call__(self, x:Tensor) -> Tensor:
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
return x.conv2d(self.weight.cast(dtypes.default_float), self.bias.cast(dtypes.default_float) if self.bias is not None else None,
|
return x.conv2d(self.weight.cast(dtypes.default_float), self.bias.cast(dtypes.default_float) if self.bias is not None else None,
|
||||||
groups=self.groups, stride=self.stride, dilation=self.dilation, padding=self.padding)
|
groups=self.groups, stride=self.stride, dilation=self.dilation, padding=self.padding)
|
||||||
|
|
||||||
|
# copy torch AMP: isolate mixed precision to just the below autocast ops, instead of using dtypes.default_float which affects all new Tensors
|
||||||
|
class AutocastLinear(nn.Linear):
|
||||||
|
cast_dtype=dtypes.bfloat16 # enable monkeypatching of the mixed precision dtype
|
||||||
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
|
dtype = type(self).cast_dtype
|
||||||
|
return x.cast(dtype).linear(self.weight.cast(dtype).transpose(), self.bias.cast(dtype) if self.bias is not None else None)
|
||||||
|
|
||||||
|
class AutocastConv2d(nn.Conv2d):
|
||||||
|
cast_dtype=dtypes.bfloat16
|
||||||
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
|
dtype = type(self).cast_dtype
|
||||||
|
return x.cast(dtype).conv2d(self.weight.cast(dtype), self.bias.cast(dtype), self.groups, self.stride, self.dilation, self.padding)
|
||||||
|
|
||||||
|
# copy torch AMP: upcast to float32 before GroupNorm and LayerNorm
|
||||||
|
class AutocastGroupNorm(nn.GroupNorm):
|
||||||
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
|
return super().__call__(x.cast(dtypes.float32))
|
||||||
|
|
||||||
|
class AutocastLayerNorm(nn.LayerNorm):
|
||||||
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
|
return super().__call__(x.cast(dtypes.float32))
|
||||||
|
|
||||||
|
def zero_module(module):
|
||||||
|
for p in get_parameters(module): p.assign(Tensor.zeros_like(p).contiguous())
|
||||||
|
|
||||||
|
# Stable Diffusion mlperf reference doesn't call scaled_dot_product_attention
|
||||||
|
# copy torch AMP: upcast to float32 before softmax on CUDA
|
||||||
|
def attn_f32_softmax(q:Tensor, k:Tensor, v:Tensor) -> Tensor:
|
||||||
|
return (q.matmul(k.transpose(-2,-1), dtype=dtypes.float32) / math.sqrt(q.shape[-1])).softmax(-1).cast(q.dtype) @ v
|
||||||
|
|
||||||
|
def init_stable_diffusion(version:str, pretrained:str, devices:list[str]):
|
||||||
|
from examples.stable_diffusion import StableDiffusion
|
||||||
|
from tinygrad.nn.state import safe_load, safe_save, load_state_dict, get_state_dict
|
||||||
|
from tempfile import TemporaryDirectory
|
||||||
|
model = StableDiffusion(version=version, pretrained=pretrained)
|
||||||
|
unet:UNetModel = model.model.diffusion_model
|
||||||
|
|
||||||
|
# this prevents extra consumption of memory, enabling much larger BS
|
||||||
|
Tensor.realize(*get_parameters(unet))
|
||||||
|
with TemporaryDirectory(prefix="unet_init") as tmp:
|
||||||
|
safe_save(get_state_dict(unet), init_fn:=f"{tmp}/init_model.safetensors")
|
||||||
|
load_state_dict(unet, safe_load(init_fn))
|
||||||
|
|
||||||
|
sqrt_alphas_cumprod = model.alphas_cumprod.sqrt().realize()
|
||||||
|
sqrt_one_minus_alphas_cumprod = (1 - model.alphas_cumprod).sqrt().realize()
|
||||||
|
|
||||||
|
if len(devices) > 1:
|
||||||
|
to_move = [sqrt_alphas_cumprod, sqrt_one_minus_alphas_cumprod]
|
||||||
|
if version == "v2-mlperf-train": to_move += get_parameters(unet) + get_parameters(model.cond_stage_model)
|
||||||
|
for p in to_move:
|
||||||
|
p.to_(devices)
|
||||||
|
with Context(BEAM=0):
|
||||||
|
Tensor.realize(*to_move)
|
||||||
|
|
||||||
|
return model, unet, sqrt_alphas_cumprod, sqrt_one_minus_alphas_cumprod
|
||||||
|
|||||||
@@ -1,8 +1,9 @@
|
|||||||
import math
|
import math
|
||||||
from tinygrad import dtypes
|
from tinygrad import dtypes, Tensor
|
||||||
from tinygrad.nn.optim import Optimizer
|
from tinygrad.nn.optim import Optimizer
|
||||||
|
|
||||||
from extra.lr_scheduler import LR_Scheduler
|
from extra.lr_scheduler import LR_Scheduler
|
||||||
|
from typing import Callable
|
||||||
|
|
||||||
# https://github.com/mlcommons/training/blob/e237206991d10449d9675d95606459a3cb6c21ad/image_classification/tensorflow2/lars_util.py
|
# https://github.com/mlcommons/training/blob/e237206991d10449d9675d95606459a3cb6c21ad/image_classification/tensorflow2/lars_util.py
|
||||||
class PolynomialDecayWithWarmup(LR_Scheduler):
|
class PolynomialDecayWithWarmup(LR_Scheduler):
|
||||||
@@ -36,4 +37,24 @@ class CosineAnnealingLRWithWarmup(LR_Scheduler):
|
|||||||
def get_lr(self):
|
def get_lr(self):
|
||||||
warmup_lr = ((self.epoch_counter+1) / self.warmup_steps) * self.base_lr
|
warmup_lr = ((self.epoch_counter+1) / self.warmup_steps) * self.base_lr
|
||||||
decay_lr = self.end_lr + 0.5 * (self.base_lr-self.end_lr) * (1 + (((self.epoch_counter+1-self.warmup_steps)/self.decay_steps) * math.pi).cos())
|
decay_lr = self.end_lr + 0.5 * (self.base_lr-self.end_lr) * (1 + (((self.epoch_counter+1-self.warmup_steps)/self.decay_steps) * math.pi).cos())
|
||||||
return (self.epoch_counter < self.warmup_steps).where(warmup_lr, decay_lr).cast(self.optimizer.lr.dtype)
|
return (self.epoch_counter < self.warmup_steps).where(warmup_lr, decay_lr).cast(self.optimizer.lr.dtype)
|
||||||
|
|
||||||
|
# Reference: https://github.com/mlcommons/training/blob/64b14a9abc74e08779a175abca7d291f8c957632/stable_diffusion/ldm/lr_scheduler.py, Lines 36-97
|
||||||
|
class LambdaLinearScheduler:
|
||||||
|
def __init__(self, warm_up_steps:int, f_min:float, f_max:float, f_start:float, cycle_lengths:int):
|
||||||
|
self.lr_warm_up_steps, self.f_min, self.f_max, self.f_start, self.cycle_lengths = warm_up_steps, f_min, f_max, f_start, cycle_lengths
|
||||||
|
|
||||||
|
def schedule(self, n:Tensor) -> Tensor:
|
||||||
|
warm_up = (n < self.lr_warm_up_steps)
|
||||||
|
f_warm_up = (self.f_max - self.f_start) / self.lr_warm_up_steps * n + self.f_start
|
||||||
|
return warm_up.where(f_warm_up, self.f_min + (self.f_max - self.f_min) * (self.cycle_lengths - n) / (self.cycle_lengths))
|
||||||
|
|
||||||
|
# based on torch.optim.lr_scheduler.LambdaLR
|
||||||
|
class LambdaLR(LR_Scheduler):
|
||||||
|
def __init__(self, optimizer:Optimizer, base_lr:Tensor, lr_lambda:Callable):
|
||||||
|
super().__init__(optimizer)
|
||||||
|
self.base_lr, self.lr_lambda = base_lr, lr_lambda
|
||||||
|
self.step()
|
||||||
|
|
||||||
|
def get_lr(self):
|
||||||
|
return self.base_lr * self.lr_lambda(self.epoch_counter - 1)
|
||||||
+280
-12
@@ -1,10 +1,10 @@
|
|||||||
import time, math
|
import time, math, os
|
||||||
start = time.perf_counter()
|
start = time.perf_counter()
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from tinygrad import Tensor, Device, dtypes, GlobalCounters, TinyJit
|
from tinygrad import Tensor, Device, dtypes, GlobalCounters, TinyJit
|
||||||
from tinygrad.nn.state import get_parameters, load_state_dict, safe_load
|
from tinygrad.nn.state import get_parameters, load_state_dict, safe_load
|
||||||
from tinygrad.helpers import getenv
|
from tinygrad.helpers import getenv, Context, prod
|
||||||
from extra.bench_log import BenchEvent, WallTimeEvent
|
from extra.bench_log import BenchEvent, WallTimeEvent
|
||||||
def tlog(x): print(f"{x:25s} @ {time.perf_counter()-start:5.2f}s")
|
def tlog(x): print(f"{x:25s} @ {time.perf_counter()-start:5.2f}s")
|
||||||
|
|
||||||
@@ -243,31 +243,299 @@ def eval_mrcnn():
|
|||||||
|
|
||||||
def eval_llama3():
|
def eval_llama3():
|
||||||
from extra.models.llama import Transformer
|
from extra.models.llama import Transformer
|
||||||
from examples.llama3 import MODEL_PARAMS
|
from examples.llama3 import MODEL_PARAMS, load, convert_from_huggingface
|
||||||
from tinygrad.helpers import tqdm
|
from tinygrad.helpers import tqdm
|
||||||
|
|
||||||
bs = 4
|
BASEDIR = Path(getenv("BASEDIR", "/raid/datasets/c4/"))
|
||||||
sequence_length = 512
|
BS = getenv("BS", 4)
|
||||||
|
SMALL = getenv("SMALL", 0)
|
||||||
|
SEQLEN = getenv("SEQLEN", 8192)
|
||||||
|
MODEL_PATH = Path(getenv("MODEL_PATH", "/raid/weights/llama31_8b/"))
|
||||||
|
|
||||||
model = Transformer(**(MODEL_PARAMS[getenv("LLAMA3_SIZE", "8B")]["args"]|{"vocab_size": 32000}), max_context=sequence_length, jit=False, disable_kv_cache=True)
|
params = MODEL_PARAMS[getenv("LLAMA3_SIZE", "8B")]["args"]
|
||||||
|
params = params | {"vocab_size": 32000} if not SMALL else params
|
||||||
|
if (llama_layers:=getenv("LLAMA_LAYERS")) != 0: params['n_layers'] = llama_layers
|
||||||
|
model = Transformer(**params, max_context=SEQLEN, jit=False, disable_kv_cache=True)
|
||||||
|
|
||||||
|
# load weights
|
||||||
|
weights = load(str(MODEL_PATH / "model.safetensors.index.json"))
|
||||||
|
if "model.embed_tokens.weight" in weights:
|
||||||
|
print("converting from huggingface format")
|
||||||
|
weights = convert_from_huggingface(weights, params["n_layers"], params["n_heads"], params["n_kv_heads"])
|
||||||
|
|
||||||
|
load_state_dict(model, weights, strict=False, consume=True)
|
||||||
|
|
||||||
@TinyJit
|
@TinyJit
|
||||||
def eval_step(model, tokens):
|
def eval_step(model, tokens):
|
||||||
logits:Tensor = model(tokens[:, :-1], start_pos=0, temperature=math.nan)
|
logits:Tensor = model(tokens[:, :-1], start_pos=0, temperature=math.nan)
|
||||||
loss = logits.sparse_categorical_crossentropy(tokens[:, 1:])
|
loss = logits.sparse_categorical_crossentropy(tokens[:, 1:])
|
||||||
return loss.flatten()
|
return loss.flatten().float()
|
||||||
|
|
||||||
from examples.mlperf.dataloader import batch_load_llama3
|
if SMALL:
|
||||||
iter = batch_load_llama3(bs, 5760, sequence_length, Path(getenv("BASEDIR", "/raid/datasets/c4/")), True)
|
from examples.mlperf.dataloader import batch_load_llama3_small
|
||||||
|
iter = batch_load_llama3_small(BS, 5760, SEQLEN, BASEDIR, val=True)
|
||||||
|
else:
|
||||||
|
from examples.mlperf.dataloader import batch_load_llama3
|
||||||
|
iter = batch_load_llama3(BS, 5760, SEQLEN, BASEDIR, val=True)
|
||||||
|
|
||||||
losses = []
|
losses = []
|
||||||
for tokens in tqdm(iter, total=5760//bs):
|
for tokens in tqdm(iter, total=5760//BS):
|
||||||
GlobalCounters.reset()
|
GlobalCounters.reset()
|
||||||
losses += eval_step(model, tokens).tolist()
|
losses += eval_step(model, tokens).tolist()
|
||||||
tqdm.write(f"loss: {np.mean(losses)}")
|
tqdm.write(f"loss: {np.mean(losses)}")
|
||||||
|
|
||||||
log_perplexity = Tensor(losses).mean()
|
log_perplexity = np.mean(losses)
|
||||||
print(f"Log Perplexity: {log_perplexity.item()}")
|
print(f"Log Perplexity: {log_perplexity}")
|
||||||
|
|
||||||
|
# NOTE: BEAM hangs on 8xmi300x with DECODE_BS=384 in final realize below; function is declared here for external testing
|
||||||
|
@TinyJit
|
||||||
|
def vae_decode(x:Tensor, vae, disable_beam=False) -> Tensor:
|
||||||
|
from examples.stable_diffusion import AutoencoderKL
|
||||||
|
assert isinstance(vae, AutoencoderKL)
|
||||||
|
x = vae.post_quant_conv(1./0.18215 * x)
|
||||||
|
|
||||||
|
x = vae.decoder.conv_in(x)
|
||||||
|
x = vae.decoder.mid(x)
|
||||||
|
for i, l in enumerate(vae.decoder.up[::-1]):
|
||||||
|
print("decode", x.shape)
|
||||||
|
for b in l['block']: x = b(x)
|
||||||
|
if 'upsample' in l:
|
||||||
|
bs,c,py,px = x.shape
|
||||||
|
x = x.reshape(bs, c, py, 1, px, 1).expand(bs, c, py, 2, px, 2).reshape(bs, c, py*2, px*2)
|
||||||
|
x = l['upsample']['conv'](x)
|
||||||
|
if i == len(vae.decoder.up) - 1 and disable_beam:
|
||||||
|
with Context(BEAM=0): x.realize()
|
||||||
|
else: x.realize()
|
||||||
|
x = vae.decoder.conv_out(vae.decoder.norm_out(x).swish())
|
||||||
|
|
||||||
|
x = ((x + 1.0) / 2.0).clip(0.0, 1.0)
|
||||||
|
return x
|
||||||
|
|
||||||
|
def eval_stable_diffusion():
|
||||||
|
import csv, PIL, sys
|
||||||
|
from tqdm import tqdm
|
||||||
|
from examples.mlperf.initializers import init_stable_diffusion, gelu_erf
|
||||||
|
from examples.stable_diffusion import AutoencoderKL
|
||||||
|
from extra.models.unet import UNetModel
|
||||||
|
from tinygrad.nn.state import load_state_dict, torch_load
|
||||||
|
from tinygrad.helpers import BEAM
|
||||||
|
from extra.models import clip
|
||||||
|
from extra.models.clip import FrozenOpenClipEmbedder
|
||||||
|
from extra.models.clip import OpenClipEncoder
|
||||||
|
from extra.models.inception import FidInceptionV3
|
||||||
|
|
||||||
|
config = {}
|
||||||
|
GPUS = config["GPUS"] = [f"{Device.DEFAULT}:{i}" for i in range(getenv("GPUS", 1))]
|
||||||
|
for x in GPUS: Device[x]
|
||||||
|
print(f"running eval on {GPUS}")
|
||||||
|
seed = config["seed"] = getenv("SEED", 12345)
|
||||||
|
CKPTDIR = config["CKPTDIR"] = Path(getenv("CKPTDIR", "./checkpoints"))
|
||||||
|
DATADIR = config["DATADIR"] = Path(getenv("DATADIR", "./datasets"))
|
||||||
|
CONTEXT_BS = config["CONTEXT_BS"] = getenv("CONTEXT_BS", 1 * len(GPUS))
|
||||||
|
DENOISE_BS = config["DENOISE_BS"] = getenv("DENOISE_BS", 1 * len(GPUS))
|
||||||
|
DECODE_BS = config["DECODE_BS"] = getenv("DECODE_BS", 1 * len(GPUS))
|
||||||
|
INCEPTION_BS = config["INCEPTION_BS"] = getenv("INCEPTION_BS", 1 * len(GPUS))
|
||||||
|
CLIP_BS = config["CLIP_BS"] = getenv("CLIP_BS", 1 * len(GPUS))
|
||||||
|
EVAL_CKPT_DIR = config["EVAL_CKPT_DIR"] = getenv("EVAL_CKPT_DIR", "")
|
||||||
|
STOP_IF_CONVERGED = config["STOP_IF_CONVERGED"] = getenv("STOP_IF_CONVERGED", 0)
|
||||||
|
|
||||||
|
if (WANDB := getenv("WANDB", "")):
|
||||||
|
import wandb
|
||||||
|
wandb.init(config=config, project="MLPerf-Stable-Diffusion")
|
||||||
|
|
||||||
|
assert EVAL_CKPT_DIR != "", "provide a directory with checkpoints to be evaluated"
|
||||||
|
print(f"running eval on checkpoints in {EVAL_CKPT_DIR}\nSEED={seed}")
|
||||||
|
eval_queue:list[tuple[int, Path]] = []
|
||||||
|
for p in Path(EVAL_CKPT_DIR).iterdir():
|
||||||
|
if p.name.endswith(".safetensors"):
|
||||||
|
ckpt_iteration = p.name.split(".safetensors")[0]
|
||||||
|
assert ckpt_iteration.isdigit(), f"invalid checkpoint name: {p.name}, expected <digits>.safetensors"
|
||||||
|
eval_queue.append((int(ckpt_iteration), p))
|
||||||
|
assert len(eval_queue), f'no files ending with ".safetensors" were found in {EVAL_CKPT_DIR}'
|
||||||
|
print(sorted(eval_queue, reverse=True))
|
||||||
|
|
||||||
|
Tensor.manual_seed(seed) # seed for weight initialization
|
||||||
|
model, unet, sqrt_alphas_cumprod, sqrt_one_minus_alphas_cumprod = init_stable_diffusion("v2-mlperf-eval", CKPTDIR / "sd" / "512-base-ema.ckpt", GPUS)
|
||||||
|
|
||||||
|
# load prompts for generating images for validation; 2 MB of data total
|
||||||
|
with open(DATADIR / "coco2014" / "val2014_30k.tsv") as f:
|
||||||
|
reader = csv.DictReader(f, delimiter="\t")
|
||||||
|
eval_inputs:list[dict] = [{"image_id": int(row["image_id"]), "id": int(row["id"]), "caption": row["caption"]} for row in reader]
|
||||||
|
assert len(eval_inputs) == 30_000
|
||||||
|
# NOTE: the clip weights are the same between model.cond_stage_model and clip_encoder
|
||||||
|
eval_timesteps = list(reversed(range(1, 1000, 20)))
|
||||||
|
|
||||||
|
original_device, Device.DEFAULT = Device.DEFAULT, "CPU"
|
||||||
|
# The choice of alphas_prev[0] = alphas_cumprod[0] seems arbitrary, but it's how the mlperf ref does it:
|
||||||
|
# alphas_prev = np.asarray([alphacums[0]] + alphacums[ddim_timesteps[:-1]].tolist())
|
||||||
|
eval_alphas_prev = model.alphas_cumprod[0:1].cat(model.alphas_cumprod[list(range(1, 1000, 20))[:-1]]).to(GPUS).realize()
|
||||||
|
inception = FidInceptionV3().load_from_pretrained(CKPTDIR / "inception" / "pt_inception-2015-12-05-6726825d.pth")
|
||||||
|
vision_cfg = {'width': 1280, 'layers': 32, 'd_head': 80, 'image_size': 224, 'patch_size': 14}
|
||||||
|
text_cfg = {'width': 1024, 'n_heads': 16, 'layers': 24, 'vocab_size': 49408, 'ctx_length': 77}
|
||||||
|
clip.gelu = gelu_erf
|
||||||
|
clip_encoder = OpenClipEncoder(1024, text_cfg, vision_cfg)
|
||||||
|
loaded = torch_load(CKPTDIR / "clip" / "open_clip_pytorch_model.bin")
|
||||||
|
loaded.update({"attn_mask": clip_encoder.attn_mask, "mean": clip_encoder.mean, "std": clip_encoder.std})
|
||||||
|
load_state_dict(clip_encoder, loaded)
|
||||||
|
Device.DEFAULT=original_device
|
||||||
|
|
||||||
|
@TinyJit
|
||||||
|
def denoise_step(x:Tensor, x_x:Tensor, t_t:Tensor, uc_c:Tensor, sqrt_alphas_cumprod_t:Tensor, sqrt_one_minus_alphas_cumprod_t:Tensor,
|
||||||
|
alpha_prev:Tensor, unet:UNetModel, GPUS) -> Tensor:
|
||||||
|
out_uncond, out = unet(x_x, t_t, uc_c).to("CPU").reshape(-1, 2, 4, 64, 64).chunk(2, dim=1)
|
||||||
|
out_uncond = out_uncond.squeeze(1).shard(GPUS,axis=0)
|
||||||
|
out = out.squeeze(1).shard(GPUS,axis=0)
|
||||||
|
v_t = out_uncond + 8.0 * (out - out_uncond)
|
||||||
|
e_t = sqrt_alphas_cumprod_t * v_t + sqrt_one_minus_alphas_cumprod_t * x
|
||||||
|
pred_x0 = sqrt_alphas_cumprod_t * x - sqrt_one_minus_alphas_cumprod_t * v_t
|
||||||
|
dir_xt = (1. - alpha_prev).sqrt() * e_t
|
||||||
|
x_prev = alpha_prev.sqrt() * pred_x0 + dir_xt
|
||||||
|
return x_prev.realize()
|
||||||
|
|
||||||
|
def shard_tensor(t:Tensor) -> Tensor: return t.shard(GPUS, axis=0) if len(GPUS) > 1 else t.to(GPUS[0])
|
||||||
|
def get_batch(whole:Tensor, i:int, bs:int) -> tuple[Tensor, int]:
|
||||||
|
batch = whole[i: i + bs].to("CPU")
|
||||||
|
if (unpadded_bs:=batch.shape[0]) < bs:
|
||||||
|
batch = batch.cat(batch[-1:].expand(bs - unpadded_bs, *batch[-1].shape))
|
||||||
|
return batch, unpadded_bs
|
||||||
|
|
||||||
|
@Tensor.train(mode=False)
|
||||||
|
def eval_unet(eval_inputs:list[dict], unet:UNetModel, cond_stage:FrozenOpenClipEmbedder, first_stage:AutoencoderKL,
|
||||||
|
inception:FidInceptionV3, clip:OpenClipEncoder) -> tuple[float, float]:
|
||||||
|
# Eval is divided into 5 jits, one per model
|
||||||
|
# It doesn't make sense to merge these jits, e.g. unet repeats 50 times in isolation; images fork to separate inception/clip
|
||||||
|
# We're generating and scoring 30,000 images per eval, and all the data can flow through one jit at a time
|
||||||
|
# To maximize throughput for each jit, we have only one model/jit on the GPU at a time, and pool outputs from each jit off-GPU
|
||||||
|
for model in (unet, first_stage, inception, clip):
|
||||||
|
Tensor.realize(*[p.to_("CPU") for p in get_parameters(model)])
|
||||||
|
|
||||||
|
uc_written = False
|
||||||
|
models = (cond_stage, unet, first_stage, inception, clip)
|
||||||
|
jits = (jit_context:=TinyJit(cond_stage.embed_tokens), denoise_step, vae_decode, jit_inception:=TinyJit(inception),
|
||||||
|
jit_clip:=TinyJit(clip.get_clip_score))
|
||||||
|
all_bs = (CONTEXT_BS, DENOISE_BS, DECODE_BS, INCEPTION_BS, CLIP_BS)
|
||||||
|
if (EVAL_SAMPLES:=getenv("EVAL_SAMPLES", 0)) and EVAL_SAMPLES > 0:
|
||||||
|
eval_inputs = eval_inputs[0:EVAL_SAMPLES]
|
||||||
|
output_shapes = [(ns:=len(eval_inputs),77), (ns,77,1024), (ns,4,64,64), (ns,3,512,512), (ns,2048), (ns,)]
|
||||||
|
# Writing progress to disk lets us resume eval if we crash
|
||||||
|
stages = ["tokens", "embeds", "latents", "imgs", "inception", "clip"]
|
||||||
|
disk_tensor_names, disk_tensor_shapes = stages + ["end", "uc"], output_shapes + [(6,), (1,77,1024)]
|
||||||
|
if not all(os.path.exists(f"{EVAL_CKPT_DIR}/{name}.bytes") for name in disk_tensor_names):
|
||||||
|
for name, shape in zip(disk_tensor_names, disk_tensor_shapes):
|
||||||
|
file = Path(f"{EVAL_CKPT_DIR}/{name}.bytes")
|
||||||
|
file.unlink(missing_ok=True)
|
||||||
|
with file.open("wb") as f: f.truncate(prod(shape) * 4)
|
||||||
|
progress = {name: Tensor.empty(*shape, device=f"disk:{EVAL_CKPT_DIR}/{name}.bytes", dtype=dtypes.int if name in {"tokens", "end"} else dtypes.float)
|
||||||
|
for name, shape in zip(disk_tensor_names, disk_tensor_shapes)}
|
||||||
|
|
||||||
|
def embed_tokens(tokens:Tensor) -> Tensor:
|
||||||
|
nonlocal uc_written
|
||||||
|
if not uc_written:
|
||||||
|
with Context(BEAM=0): progress["uc"].assign(cond_stage.embed_tokens(cond_stage.tokenize("").to(GPUS)).to("CPU").realize()).realize()
|
||||||
|
uc_written = True
|
||||||
|
return jit_context(shard_tensor(tokens))
|
||||||
|
|
||||||
|
def generate_latents(embeds:Tensor) -> Tensor:
|
||||||
|
uc_c = Tensor.stack(progress["uc"].to("CPU").expand(bs, 77, 1024), embeds, dim=1).reshape(-1, 77, 1024)
|
||||||
|
uc_c = shard_tensor(uc_c)
|
||||||
|
x = shard_tensor(Tensor.randn(bs,4,64,64))
|
||||||
|
for step_idx, timestep in enumerate(tqdm(eval_timesteps)):
|
||||||
|
reversed_idx = Tensor([50 - step_idx - 1], device=GPUS)
|
||||||
|
alpha_prev = eval_alphas_prev[reversed_idx]
|
||||||
|
ts = Tensor.full(bs, fill_value=timestep, dtype=dtypes.int, device="CPU")
|
||||||
|
ts_ts = shard_tensor(ts.cat(ts))
|
||||||
|
ts = shard_tensor(ts)
|
||||||
|
sqrt_alphas_cumprod_t = sqrt_alphas_cumprod[ts].reshape(bs, 1, 1, 1)
|
||||||
|
sqrt_one_minus_alphas_cumprod_t = sqrt_one_minus_alphas_cumprod[ts].reshape(bs, 1, 1, 1)
|
||||||
|
x_x = shard_tensor(Tensor.stack(x.to("CPU"), x.to("CPU"), dim=1).reshape(-1, 4, 64, 64))
|
||||||
|
x.assign(denoise_step(x, x_x, ts_ts, uc_c, sqrt_alphas_cumprod_t, sqrt_one_minus_alphas_cumprod_t, alpha_prev, unet, GPUS)).realize()
|
||||||
|
return x
|
||||||
|
|
||||||
|
def decode_latents(latents:Tensor) -> Tensor: return vae_decode(shard_tensor(latents), first_stage, disable_beam=True)
|
||||||
|
def generate_inception(imgs:Tensor) -> Tensor: return jit_inception(shard_tensor(imgs))[:,:,0,0]
|
||||||
|
|
||||||
|
def calc_clip_scores(batch:Tensor, batch_tokens:Tensor) -> Tensor:
|
||||||
|
# Tensor.interpolate does not yet support bicubic, so we use PIL
|
||||||
|
batch = (batch.to(GPUS[0]).permute(0,2,3,1) * 255).clip(0, 255).cast(dtypes.uint8).numpy()
|
||||||
|
batch = [np.array(PIL.Image.fromarray(batch[i]).resize((224,224), PIL.Image.BICUBIC)) for i in range(bs)]
|
||||||
|
batch = shard_tensor(Tensor(np.stack(batch, axis=0).transpose(0,3,1,2), device="CPU").realize())
|
||||||
|
batch = batch.cast(dtypes.float) / 255
|
||||||
|
batch = (batch - model.mean) / model.std
|
||||||
|
batch = jit_clip(shard_tensor(batch_tokens), batch)
|
||||||
|
return batch
|
||||||
|
|
||||||
|
callbacks = (embed_tokens, generate_latents, decode_latents, generate_inception, calc_clip_scores)
|
||||||
|
|
||||||
|
# save every forward pass output to disk; NOTE: this needs ~100 GB disk space because 30k images are large
|
||||||
|
def stage_progress(stage_idx:int) -> int: return progress["end"].to("CPU")[stage_idx].item()
|
||||||
|
if stage_progress(0) < len(eval_inputs):
|
||||||
|
tokens = []
|
||||||
|
for i in tqdm(range(0, len(eval_inputs), CONTEXT_BS)):
|
||||||
|
subset = [cond_stage.tokenize(row["caption"], device="CPU") for row in eval_inputs[i: i+CONTEXT_BS]]
|
||||||
|
tokens.append(Tensor.cat(*subset, dim=0).realize())
|
||||||
|
progress["tokens"].assign(Tensor.cat(*tokens, dim=0).realize()).realize()
|
||||||
|
progress["end"][0:1].assign(Tensor([len(eval_inputs)], dtype=dtypes.int)).realize()
|
||||||
|
prev_stage = "tokens"
|
||||||
|
tokens = progress["tokens"]
|
||||||
|
|
||||||
|
# wrapper code for every model
|
||||||
|
for stage_idx, model, jit, bs, callback in zip(range(1,6), models, jits, all_bs, callbacks):
|
||||||
|
stage = stages[stage_idx]
|
||||||
|
if stage_progress(stage_idx) >= len(eval_inputs):
|
||||||
|
prev_stage = stage
|
||||||
|
continue # use cache
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
print(f"starting eval with model: {model}")
|
||||||
|
if stage_idx == 1: inputs = tokens
|
||||||
|
elif stage_idx == 5: inputs = progress["imgs"]
|
||||||
|
else: inputs = progress[prev_stage]
|
||||||
|
|
||||||
|
Tensor.realize(*[p.to_(GPUS) for p in get_parameters(model)])
|
||||||
|
for batch_idx in tqdm(range(stage_progress(stage_idx), inputs.shape[0], bs)):
|
||||||
|
t1 = time.perf_counter()
|
||||||
|
batch, unpadded_bs = get_batch(inputs, batch_idx, bs)
|
||||||
|
if isinstance(model, OpenClipEncoder): batch = callback(batch, get_batch(tokens, batch_idx, bs)[0].realize())
|
||||||
|
else: batch = callback(batch)
|
||||||
|
# to(GPUS[0]) is necessary for this to work, without that the result is still on GPUS, probably due to a bug
|
||||||
|
batch = batch.to(GPUS[0]).to("CPU")[0:unpadded_bs].realize()
|
||||||
|
progress[stage][batch_idx: batch_idx + bs].assign(batch).realize()
|
||||||
|
# keep track of what our last output was, so we can resume from there if we crash in this loop
|
||||||
|
progress["end"][stage_idx: stage_idx + 1].assign(Tensor([batch_idx + bs], dtype=dtypes.int)).realize()
|
||||||
|
print(f"model: {model}, batch_idx: {batch_idx}, elapsed: {(time.perf_counter() - t1):.2f}")
|
||||||
|
del batch
|
||||||
|
|
||||||
|
jit.reset()
|
||||||
|
Tensor.realize(*[p.to_("CPU") for p in get_parameters(model)])
|
||||||
|
print(f"done with model: {model}, elapsed: {(time.perf_counter() - t0):.2f}")
|
||||||
|
prev_stage = stage
|
||||||
|
|
||||||
|
inception_stats_fn = str(DATADIR / "coco2014" / "val2014_30k_stats.npz")
|
||||||
|
fid_score = inception.compute_score(progress["inception"].to("CPU"), inception_stats_fn)
|
||||||
|
clip_score = progress["clip"].to(GPUS[0]).mean().item()
|
||||||
|
for name in disk_tensor_names:
|
||||||
|
Path(f"{EVAL_CKPT_DIR}/{name}.bytes").unlink(missing_ok=True)
|
||||||
|
|
||||||
|
if EVAL_SAMPLES and BEAM:
|
||||||
|
print("BEAM COMPLETE", flush=True) # allows wrapper script to detect BEAM search completion and retry if it failed
|
||||||
|
sys.exit() # Don't eval additional models; we don't care about clip/fid scores when running BEAM on eval sample subset
|
||||||
|
|
||||||
|
return clip_score, fid_score
|
||||||
|
|
||||||
|
# evaluate checkpoints in reverse chronological order
|
||||||
|
for ckpt_iteration, p in sorted(eval_queue, reverse=True):
|
||||||
|
unet_ckpt = safe_load(p)
|
||||||
|
load_state_dict(unet, unet_ckpt)
|
||||||
|
clip_score, fid_score = eval_unet(eval_inputs, unet, model.cond_stage_model, model.first_stage_model, inception, clip_encoder)
|
||||||
|
converged = True if clip_score >= 0.15 and fid_score <= 90 else False
|
||||||
|
print(f"eval results for {EVAL_CKPT_DIR}/{p.name}: clip={clip_score}, fid={fid_score}, converged={converged}")
|
||||||
|
if WANDB:
|
||||||
|
wandb.log({"eval/ckpt_iteration": ckpt_iteration, "eval/clip_score": clip_score, "eval/fid_score": fid_score})
|
||||||
|
if converged and STOP_IF_CONVERGED:
|
||||||
|
print(f"Convergence detected, exiting early before evaluating other checkpoints due to STOP_IF_CONVERGED={STOP_IF_CONVERGED}")
|
||||||
|
sys.exit()
|
||||||
|
|
||||||
|
# for testing
|
||||||
|
return clip_score, fid_score, ckpt_iteration
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
# inference only
|
# inference only
|
||||||
|
|||||||
+190
-21
@@ -3,8 +3,8 @@ from pathlib import Path
|
|||||||
import multiprocessing
|
import multiprocessing
|
||||||
|
|
||||||
from tinygrad import Device, GlobalCounters, Tensor, TinyJit, dtypes
|
from tinygrad import Device, GlobalCounters, Tensor, TinyJit, dtypes
|
||||||
from tinygrad.helpers import getenv, BEAM, WINO, round_up, diskcache_clear, FUSE_CONV_BW, Profiling
|
from tinygrad.helpers import getenv, BEAM, WINO, round_up, diskcache_clear, Profiling
|
||||||
from tinygrad.nn.state import get_parameters, get_state_dict, safe_load, safe_save
|
from tinygrad.nn.state import get_parameters, get_state_dict, load_state_dict, safe_load, safe_save
|
||||||
from tinygrad.nn.optim import LAMB, LARS, SGD, OptimizerGroup, Adam, AdamW
|
from tinygrad.nn.optim import LAMB, LARS, SGD, OptimizerGroup, Adam, AdamW
|
||||||
|
|
||||||
from extra.lr_scheduler import LRSchedulerGroup
|
from extra.lr_scheduler import LRSchedulerGroup
|
||||||
@@ -252,6 +252,10 @@ def train_resnet():
|
|||||||
print(f"epoch global_ops: {steps_in_train_epoch * GlobalCounters.global_ops:_}, "
|
print(f"epoch global_ops: {steps_in_train_epoch * GlobalCounters.global_ops:_}, "
|
||||||
f"epoch global_mem: {steps_in_train_epoch * GlobalCounters.global_mem:_}")
|
f"epoch global_mem: {steps_in_train_epoch * GlobalCounters.global_mem:_}")
|
||||||
# if we are doing beam search, run the first eval too
|
# if we are doing beam search, run the first eval too
|
||||||
|
if (assert_time:=getenv("ASSERT_MIN_STEP_TIME")):
|
||||||
|
min_time = min(step_times)
|
||||||
|
assert min_time < assert_time, f"Speed regression, expected min step time of < {assert_time} ms but took: {min_time} ms"
|
||||||
|
|
||||||
if (TRAIN_BEAM or EVAL_BEAM) and e == start_epoch: break
|
if (TRAIN_BEAM or EVAL_BEAM) and e == start_epoch: break
|
||||||
return
|
return
|
||||||
if MLLOGGER and RUNMLPERF:
|
if MLLOGGER and RUNMLPERF:
|
||||||
@@ -344,6 +348,8 @@ def train_resnet():
|
|||||||
print(f"saving ckpt to {fn}")
|
print(f"saving ckpt to {fn}")
|
||||||
safe_save(get_training_state(model, optimizer_group, scheduler_group), fn)
|
safe_save(get_training_state(model, optimizer_group, scheduler_group), fn)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def train_retinanet():
|
def train_retinanet():
|
||||||
from contextlib import redirect_stdout
|
from contextlib import redirect_stdout
|
||||||
from examples.mlperf.dataloader import batch_load_retinanet
|
from examples.mlperf.dataloader import batch_load_retinanet
|
||||||
@@ -701,7 +707,7 @@ def train_unet3d():
|
|||||||
```BASEDIR=<folder_path> ./examples/mlperf/scripts/setup_kits19_dataset.sh```
|
```BASEDIR=<folder_path> ./examples/mlperf/scripts/setup_kits19_dataset.sh```
|
||||||
|
|
||||||
2) To start training the model, run the following:
|
2) To start training the model, run the following:
|
||||||
```time PYTHONPATH=. WANDB=1 TRAIN_BEAM=3 FUSE_CONV_BW=1 GPUS=6 BS=6 MODEL=unet3d python3 examples/mlperf/model_train.py```
|
```time PYTHONPATH=. WANDB=1 TRAIN_BEAM=3 GPUS=6 BS=6 MODEL=unet3d python3 examples/mlperf/model_train.py```
|
||||||
"""
|
"""
|
||||||
from examples.mlperf.losses import dice_ce_loss
|
from examples.mlperf.losses import dice_ce_loss
|
||||||
from examples.mlperf.metrics import dice_score
|
from examples.mlperf.metrics import dice_score
|
||||||
@@ -743,7 +749,6 @@ def train_unet3d():
|
|||||||
"train_beam": TRAIN_BEAM,
|
"train_beam": TRAIN_BEAM,
|
||||||
"eval_beam": EVAL_BEAM,
|
"eval_beam": EVAL_BEAM,
|
||||||
"wino": WINO.value,
|
"wino": WINO.value,
|
||||||
"fuse_conv_bw": FUSE_CONV_BW.value,
|
|
||||||
"gpus": GPUS,
|
"gpus": GPUS,
|
||||||
"default_float": dtypes.default_float.name
|
"default_float": dtypes.default_float.name
|
||||||
}
|
}
|
||||||
@@ -1183,7 +1188,9 @@ def train_bert():
|
|||||||
if MLLOGGER and RUNMLPERF:
|
if MLLOGGER and RUNMLPERF:
|
||||||
MLLOGGER.start(key=mllog_constants.EVAL_START, value=None, metadata={"epoch_num": i*GBS, "step_num": i})
|
MLLOGGER.start(key=mllog_constants.EVAL_START, value=None, metadata={"epoch_num": i*GBS, "step_num": i})
|
||||||
if getenv("RESET_STEP"): train_step_bert.reset()
|
if getenv("RESET_STEP"): train_step_bert.reset()
|
||||||
elif getenv("FREE_INTERMEDIATE", 1) and train_step_bert.captured is not None: train_step_bert.captured.free_intermediates()
|
elif getenv("FREE_INTERMEDIATE", 0) and train_step_bert.captured is not None:
|
||||||
|
# TODO: FREE_INTERMEDIATE nan'ed after jit step 2
|
||||||
|
train_step_bert.captured.free_intermediates()
|
||||||
eval_lm_losses = []
|
eval_lm_losses = []
|
||||||
eval_clsf_losses = []
|
eval_clsf_losses = []
|
||||||
eval_lm_accs = []
|
eval_lm_accs = []
|
||||||
@@ -1217,7 +1224,7 @@ def train_bert():
|
|||||||
return
|
return
|
||||||
|
|
||||||
if getenv("RESET_STEP"): eval_step_bert.reset()
|
if getenv("RESET_STEP"): eval_step_bert.reset()
|
||||||
elif getenv("FREE_INTERMEDIATE", 1) and eval_step_bert.captured is not None: eval_step_bert.captured.free_intermediates()
|
elif getenv("FREE_INTERMEDIATE", 0) and eval_step_bert.captured is not None: eval_step_bert.captured.free_intermediates()
|
||||||
|
|
||||||
del eval_data
|
del eval_data
|
||||||
avg_lm_loss = sum(eval_lm_losses) / len(eval_lm_losses)
|
avg_lm_loss = sum(eval_lm_losses) / len(eval_lm_losses)
|
||||||
@@ -1290,18 +1297,20 @@ def train_llama3():
|
|||||||
from examples.mlperf.lr_schedulers import CosineAnnealingLRWithWarmup
|
from examples.mlperf.lr_schedulers import CosineAnnealingLRWithWarmup
|
||||||
|
|
||||||
config = {}
|
config = {}
|
||||||
|
BASEDIR = config["BASEDIR"] = Path(getenv("BASEDIR", "/raid/datasets/c4/"))
|
||||||
BS = config["BS"] = getenv("BS", 16)
|
BS = config["BS"] = getenv("BS", 16)
|
||||||
grad_acc = config["GRADIENT_ACC_STEPS"] = getenv("GRADIENT_ACC_STEPS", 1)
|
grad_acc = config["GRADIENT_ACC_STEPS"] = getenv("GRADIENT_ACC_STEPS", 1)
|
||||||
GBS = config["GLOBAL_BATCH_SIZE"] = BS * grad_acc
|
GBS = config["GLOBAL_BATCH_SIZE"] = BS * grad_acc
|
||||||
SEED = config["SEED"] = getenv("SEED", 5760)
|
SEED = config["SEED"] = getenv("SEED", 5760)
|
||||||
SEQLEN = config["SEQLEN"] = getenv("SEQLEN", 8192)
|
SEQLEN = config["SEQLEN"] = getenv("SEQLEN", 8192)
|
||||||
TRAIN_ON_VAL = config["TRAIN_ON_VAL"] = getenv("TRAIN_ON_VAL", 0)
|
TRAIN_ON_VAL = config["TRAIN_ON_VAL"] = getenv("TRAIN_ON_VAL", 0)
|
||||||
|
SMALL = config["SMALL"] = getenv("SMALL", 0)
|
||||||
SAMPLES = config["SAMPLES"] = getenv("SAMPLES", 5_760 if TRAIN_ON_VAL else 1_200_000 * 1152)
|
SAMPLES = config["SAMPLES"] = getenv("SAMPLES", 5_760 if TRAIN_ON_VAL else 1_200_000 * 1152)
|
||||||
EVAL_FREQ = config["EVAL_FREQ"] = getenv("EVAL_FREQ", 46080)
|
EVAL_FREQ = config["EVAL_FREQ"] = getenv("EVAL_FREQ", 46080)
|
||||||
EVAL_BS = config["EVAL_BS"] = getenv("EVAL_BS", 16)
|
EVAL_BS = config["EVAL_BS"] = getenv("EVAL_BS", 16)
|
||||||
EVAL_TARGET = config["EVAL_TARGET"] = getenv("EVAL_TARGET", 5.6)
|
EVAL_TARGET = config["EVAL_TARGET"] = getenv("EVAL_TARGET", 5.6)
|
||||||
|
|
||||||
# LR=1e-4 TRAIN_ON_VAL=1 DEFAULT_FLOAT=bfloat16 FUSE_ARANGE=1 JITBEAM=2 OPTIM_DTYPE=bfloat16 LLAMA3_SIZE=1B WARMUP_STEPS=36 DECAY_STEPS=360 SEQLEN=512 PYTHONPATH=. AMD=1 AMD_LLVM=0 MODEL=llama3 python3 examples/mlperf/model_train.py
|
# LR=1e-4 TRAIN_ON_VAL=1 DEFAULT_FLOAT=bfloat16 JITBEAM=2 OPTIM_DTYPE=bfloat16 LLAMA3_SIZE=1B WARMUP_STEPS=36 DECAY_STEPS=360 SEQLEN=512 PYTHONPATH=. AMD=1 AMD_LLVM=0 MODEL=llama3 python3 examples/mlperf/model_train.py
|
||||||
# trains to 7
|
# trains to 7
|
||||||
|
|
||||||
opt_adamw_beta_1 = 0.9
|
opt_adamw_beta_1 = 0.9
|
||||||
@@ -1311,13 +1320,14 @@ def train_llama3():
|
|||||||
|
|
||||||
opt_gradient_clip_norm = 1.0
|
opt_gradient_clip_norm = 1.0
|
||||||
opt_learning_rate_warmup_steps = getenv("WARMUP_STEPS", math.ceil(8000 * 1152 / GBS))
|
opt_learning_rate_warmup_steps = getenv("WARMUP_STEPS", math.ceil(8000 * 1152 / GBS))
|
||||||
opt_learning_rate_decay_steps = getenv("DECAY_STEPS", math.ceil(1_200_000 * 1152 / GBS) - opt_learning_rate_warmup_steps)
|
opt_learning_rate_decay_steps = getenv("MAX_STEPS", math.ceil(1_200_000 * 1152 / GBS)) - opt_learning_rate_warmup_steps
|
||||||
opt_base_learning_rate = getenv("LR", 8e-5 * GBS / 1152) # NOTE: cannot change for benchmark
|
opt_base_learning_rate = getenv("LR", 8e-5 * GBS / 1152) # NOTE: cannot change for benchmark
|
||||||
opt_end_learning_rate = 8e-7
|
opt_end_learning_rate = getenv("END_LR", 8e-7)
|
||||||
|
|
||||||
# TODO: confirm weights are in bf16
|
# TODO: confirm weights are in bf16
|
||||||
# vocab_size from the mixtral tokenizer
|
# vocab_size from the mixtral tokenizer
|
||||||
params = MODEL_PARAMS[getenv("LLAMA3_SIZE", "8B")]["args"]|{"vocab_size": 32000}
|
params = MODEL_PARAMS[getenv("LLAMA3_SIZE", "8B")]["args"]
|
||||||
|
params = params | {"vocab_size": 32000} if not SMALL else params
|
||||||
if (llama_layers:=getenv("LLAMA_LAYERS")) != 0: params['n_layers'] = llama_layers
|
if (llama_layers:=getenv("LLAMA_LAYERS")) != 0: params['n_layers'] = llama_layers
|
||||||
model = Transformer(**params, max_context=SEQLEN, jit=False, disable_kv_cache=True)
|
model = Transformer(**params, max_context=SEQLEN, jit=False, disable_kv_cache=True)
|
||||||
|
|
||||||
@@ -1353,6 +1363,15 @@ def train_llama3():
|
|||||||
b1=opt_adamw_beta_1, b2=opt_adamw_beta_2, eps=opt_adamw_epsilon, weight_decay=opt_adamw_weight_decay)
|
b1=opt_adamw_beta_1, b2=opt_adamw_beta_2, eps=opt_adamw_epsilon, weight_decay=opt_adamw_weight_decay)
|
||||||
scheduler = CosineAnnealingLRWithWarmup(optim, opt_base_learning_rate, opt_end_learning_rate, opt_learning_rate_warmup_steps, opt_learning_rate_decay_steps)
|
scheduler = CosineAnnealingLRWithWarmup(optim, opt_base_learning_rate, opt_end_learning_rate, opt_learning_rate_warmup_steps, opt_learning_rate_decay_steps)
|
||||||
|
|
||||||
|
if resume_ckpt := getenv("RESUME_CKPT"):
|
||||||
|
fn = f"./ckpts/llama3_{resume_ckpt}.safe"
|
||||||
|
print(f"loading initial checkpoint from {fn}")
|
||||||
|
load_state_dict(model, safe_load(fn), realize=False)
|
||||||
|
|
||||||
|
fn = f"./ckpts/llama3_{resume_ckpt}_optim.safe"
|
||||||
|
print(f"loading optim checkpoint from {fn}")
|
||||||
|
load_state_dict(scheduler, safe_load(fn), realize=False)
|
||||||
|
|
||||||
@TinyJit
|
@TinyJit
|
||||||
@Tensor.train()
|
@Tensor.train()
|
||||||
def train_step(model, tokens:Tensor, grad_acc:int):
|
def train_step(model, tokens:Tensor, grad_acc:int):
|
||||||
@@ -1403,43 +1422,55 @@ def train_llama3():
|
|||||||
# ** data iters **
|
# ** data iters **
|
||||||
def fake_data(bs, samples):
|
def fake_data(bs, samples):
|
||||||
for _ in range(samples // bs):
|
for _ in range(samples // bs):
|
||||||
yield Tensor.randint(bs, SEQLEN + 1, low=0, high=32000, dtype=dtypes.int32, device=Device.DEFAULT)
|
yield Tensor.randint(bs, SEQLEN + 1, low=0, high=params["vocab_size"], dtype=dtypes.int32, device=Device.DEFAULT)
|
||||||
|
|
||||||
def get_train_iter():
|
def get_train_iter():
|
||||||
if getenv("FAKEDATA", 0):
|
if getenv("FAKEDATA", 0):
|
||||||
return fake_data(GBS, SAMPLES)
|
return fake_data(GBS, SAMPLES)
|
||||||
else:
|
else:
|
||||||
from examples.mlperf.dataloader import batch_load_llama3
|
if SMALL:
|
||||||
return batch_load_llama3(GBS, SAMPLES, SEQLEN, Path(getenv("BASEDIR", "/raid/datasets/c4/")), seed=SEED, val=bool(TRAIN_ON_VAL))
|
from examples.mlperf.dataloader import batch_load_llama3_small
|
||||||
|
return batch_load_llama3_small(GBS, SAMPLES, SEQLEN, BASEDIR, seed=SEED, val=bool(TRAIN_ON_VAL))
|
||||||
|
else:
|
||||||
|
from examples.mlperf.dataloader import batch_load_llama3
|
||||||
|
return batch_load_llama3(GBS, SAMPLES, SEQLEN, BASEDIR, seed=SEED, val=bool(TRAIN_ON_VAL))
|
||||||
|
|
||||||
def get_eval_iter():
|
def get_eval_iter():
|
||||||
if getenv("FAKEDATA", 0):
|
if getenv("FAKEDATA", 0):
|
||||||
return fake_data(EVAL_BS, 5760)
|
return fake_data(EVAL_BS, 5760)
|
||||||
else:
|
else:
|
||||||
from examples.mlperf.dataloader import batch_load_llama3
|
if SMALL:
|
||||||
return batch_load_llama3(EVAL_BS, 5760, SEQLEN, Path(getenv("BASEDIR", "/raid/datasets/c4/")), seed=SEED, val=True)
|
from examples.mlperf.dataloader import batch_load_llama3_small
|
||||||
|
return batch_load_llama3_small(EVAL_BS, 5760, SEQLEN, BASEDIR, val=True)
|
||||||
|
else:
|
||||||
|
from examples.mlperf.dataloader import batch_load_llama3
|
||||||
|
return batch_load_llama3(EVAL_BS, 5760, SEQLEN, BASEDIR, val=True)
|
||||||
|
|
||||||
iter = get_train_iter()
|
iter = get_train_iter()
|
||||||
i, sequences_seen = 0, 0
|
i, sequences_seen = resume_ckpt, 0
|
||||||
for tokens in tqdm(iter, total=SAMPLES//GBS):
|
for tokens in tqdm(iter, total=SAMPLES//GBS):
|
||||||
t = time.perf_counter()
|
t = time.perf_counter()
|
||||||
GlobalCounters.reset()
|
GlobalCounters.reset()
|
||||||
loss, lr = train_step(model, tokens, grad_acc)
|
loss, lr = train_step(model, tokens, grad_acc)
|
||||||
loss = loss.float().item()
|
loss = loss.float().item()
|
||||||
# above as tqdm.write f-string
|
|
||||||
|
i += 1
|
||||||
|
sequences_seen += tokens.shape[0]
|
||||||
|
|
||||||
tqdm.write(f"{loss:.4f} loss, {lr.item():.12f} LR, {GlobalCounters.mem_used / 1e9:.2f} GB used, {time.perf_counter()-t:.2f} s")
|
tqdm.write(f"{loss:.4f} loss, {lr.item():.12f} LR, {GlobalCounters.mem_used / 1e9:.2f} GB used, {time.perf_counter()-t:.2f} s")
|
||||||
if (fname:=getenv("LOSS_FILE", "")):
|
if (fname:=getenv("LOSS_FILE", "")):
|
||||||
with open(fname, "a") as f:
|
with open(fname, "a") as f:
|
||||||
f.write(f"{i} {loss:.4f} {lr.item():.12f} {GlobalCounters.mem_used / 1e9:.2f}\n")
|
f.write(f"{i} {loss:.4f} {lr.item():.12f} {GlobalCounters.mem_used / 1e9:.2f}\n")
|
||||||
|
|
||||||
if getenv("CKPT") and (i % 200 == 0 or i == 10):
|
if (ckpt_freq := getenv("CKPT")) and (i % ckpt_freq == 0 and (i != 1 or ckpt_freq == 1)):
|
||||||
tqdm.write("saving checkpoint")
|
tqdm.write("saving checkpoint")
|
||||||
if not os.path.exists(ckpt_dir := "./ckpts"): os.mkdir(ckpt_dir)
|
if not os.path.exists(ckpt_dir := "./ckpts"): os.mkdir(ckpt_dir)
|
||||||
fn = f"{ckpt_dir}/llama3_{i}.safe"
|
fn = f"{ckpt_dir}/llama3_{i}.safe"
|
||||||
safe_save(get_state_dict(model), fn)
|
safe_save(get_state_dict(model), fn)
|
||||||
|
|
||||||
i += 1
|
tqdm.write("saving optim checkpoint")
|
||||||
sequences_seen += tokens.shape[0]
|
fn = f"{ckpt_dir}/llama3_{i}_optim.safe"
|
||||||
|
safe_save(get_state_dict(scheduler), fn)
|
||||||
|
|
||||||
if sequences_seen % EVAL_FREQ == 0 and (i != 1 or EVAL_FREQ == 1):
|
if sequences_seen % EVAL_FREQ == 0 and (i != 1 or EVAL_FREQ == 1):
|
||||||
tqdm.write(f"evaluating after {sequences_seen} sequences")
|
tqdm.write(f"evaluating after {sequences_seen} sequences")
|
||||||
@@ -1463,6 +1494,144 @@ def train_llama3():
|
|||||||
safe_save(get_state_dict(model), fn)
|
safe_save(get_state_dict(model), fn)
|
||||||
break
|
break
|
||||||
|
|
||||||
|
def train_stable_diffusion():
|
||||||
|
from extra.models.unet import UNetModel
|
||||||
|
from examples.mlperf.dataloader import batch_load_train_stable_diffusion
|
||||||
|
from examples.mlperf.lr_schedulers import LambdaLR, LambdaLinearScheduler
|
||||||
|
from examples.mlperf.initializers import init_stable_diffusion
|
||||||
|
from examples.mlperf.helpers import get_training_state
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
config = {}
|
||||||
|
GPUS = config["GPUS"] = [f"{Device.DEFAULT}:{i}" for i in range(getenv("GPUS", 1))]
|
||||||
|
seed = config["seed"] = getenv("SEED", 12345)
|
||||||
|
# ** hyperparameters **
|
||||||
|
BS = config["BS"] = getenv("BS", 1 * len(GPUS))
|
||||||
|
BASE_LR = config["LEARNING_RATE"] = getenv("LEARNING_RATE", 2.5e-7)
|
||||||
|
# https://github.com/mlcommons/training_policies/blob/cfa99da479b8d5931f7a3c67612d021dfb47510a/training_rules.adoc#benchmark_specific_rules
|
||||||
|
# "Checkpoint must be collected every 512,000 images. CEIL(512000 / global_batch_size) if 512000 is not divisible by GBS."
|
||||||
|
# NOTE: It's inferred that "steps" is the unit for the output of the CEIL formula, based on all other cases of CEIL in the rules
|
||||||
|
CKPT_STEP_INTERVAL = config["CKPT_STEP_INTERVAL"] = getenv("CKPT_STEP_INTERVAL", math.ceil(512_000 / BS))
|
||||||
|
CKPTDIR = config["CKPTDIR"] = Path(getenv("CKPTDIR", "./checkpoints"))
|
||||||
|
DATADIR = config["DATADIR"] = Path(getenv("DATADIR", "./datasets"))
|
||||||
|
UNET_CKPTDIR = config["UNET_CKPTDIR"] = Path(getenv("UNET_CKPTDIR", "./checkpoints"))
|
||||||
|
TOTAL_CKPTS = config["TOTAL_CKPTS"] = getenv("TOTAL_CKPTS", 0)
|
||||||
|
|
||||||
|
print(f"training on {GPUS}")
|
||||||
|
lr = BS * BASE_LR
|
||||||
|
print(f"BS={BS}, BASE_LR={BASE_LR}, lr={lr}")
|
||||||
|
print(f"CKPT_STEP_INTERVAL = {CKPT_STEP_INTERVAL}")
|
||||||
|
for x in GPUS: Device[x]
|
||||||
|
if (WANDB := getenv("WANDB", "")):
|
||||||
|
import wandb
|
||||||
|
wandb.init(config=config, project="MLPerf-Stable-Diffusion")
|
||||||
|
|
||||||
|
Tensor.manual_seed(seed) # seed for weight initialization
|
||||||
|
model, unet, sqrt_alphas_cumprod, sqrt_one_minus_alphas_cumprod = init_stable_diffusion("v2-mlperf-train", CKPTDIR / "sd" / "512-base-ema.ckpt", GPUS)
|
||||||
|
|
||||||
|
optimizer = AdamW(get_parameters(unet))
|
||||||
|
lambda_lr_callback = LambdaLinearScheduler(1000, 1.0, 1.0, 1e-06, 10000000000000).schedule
|
||||||
|
lr_scheduler = LambdaLR(optimizer, Tensor(lr, dtype=dtypes.float, device=optimizer.device), lambda_lr_callback)
|
||||||
|
|
||||||
|
@TinyJit
|
||||||
|
def train_step(mean:Tensor, logvar:Tensor, tokens:Tensor, unet:UNetModel, optimizer:LAMB, lr_scheduler:LambdaLR) -> Tensor:
|
||||||
|
optimizer.zero_grad()
|
||||||
|
|
||||||
|
timestep = Tensor.randint(BS, low=0, high=model.alphas_cumprod.shape[0], dtype=dtypes.int, device=GPUS[0])
|
||||||
|
latent_randn = Tensor.randn(*mean.shape, device=GPUS[0])
|
||||||
|
noise = Tensor.randn(*mean.shape, device=GPUS[0])
|
||||||
|
for t in (mean, logvar, tokens, timestep, latent_randn, noise):
|
||||||
|
t.shard_(GPUS, axis=0)
|
||||||
|
|
||||||
|
std = Tensor.exp(0.5 * logvar.clamp(-30.0, 20.0))
|
||||||
|
latent = (mean + std * latent_randn) * 0.18215
|
||||||
|
|
||||||
|
sqrt_alphas_cumprod_t = sqrt_alphas_cumprod[timestep].reshape(timestep.shape[0], 1, 1, 1)
|
||||||
|
sqrt_one_minus_alphas_cumprod_t = sqrt_one_minus_alphas_cumprod[timestep].reshape(timestep.shape[0], 1, 1, 1)
|
||||||
|
latent_with_noise = sqrt_alphas_cumprod_t * latent + sqrt_one_minus_alphas_cumprod_t * noise
|
||||||
|
v_true = sqrt_alphas_cumprod_t * noise - sqrt_one_minus_alphas_cumprod_t * latent
|
||||||
|
|
||||||
|
context = model.cond_stage_model.embed_tokens(tokens)
|
||||||
|
|
||||||
|
out = unet(latent_with_noise, timestep, context)
|
||||||
|
loss = ((out - v_true) ** 2).mean()
|
||||||
|
del mean, logvar, std, latent, noise, sqrt_alphas_cumprod_t, sqrt_one_minus_alphas_cumprod_t
|
||||||
|
del out, v_true, context, latent_randn, tokens, timestep
|
||||||
|
loss.backward()
|
||||||
|
|
||||||
|
optimizer.step()
|
||||||
|
lr_scheduler.step()
|
||||||
|
loss, out_lr = loss.detach().to("CPU"), optimizer.lr.to("CPU")
|
||||||
|
Tensor.realize(loss, out_lr)
|
||||||
|
return loss, out_lr
|
||||||
|
|
||||||
|
# checkpointing takes ~9 minutes without this, and ~1 minute with this
|
||||||
|
@TinyJit
|
||||||
|
def ckpt_to_cpu():
|
||||||
|
ckpt = get_training_state(unet, optimizer, lr_scheduler)
|
||||||
|
# move to CPU first so more GPU bufs aren't created (can trigger OOM)
|
||||||
|
for k,v in ckpt.items(): ckpt[k] = v.detach().to("CPU")
|
||||||
|
Tensor.realize(*[v for v in ckpt.values()])
|
||||||
|
for k,v in ckpt.items(): ckpt[k] = v.cast(v.dtype.base).contiguous()
|
||||||
|
Tensor.realize(*[v for v in ckpt.values()])
|
||||||
|
return ckpt
|
||||||
|
|
||||||
|
# training loop
|
||||||
|
dl = batch_load_train_stable_diffusion(f'{DATADIR}/laion-400m/webdataset-moments-filtered/{{00000..00831}}.tar', BS)
|
||||||
|
# for tests
|
||||||
|
saved_checkpoints = []
|
||||||
|
|
||||||
|
train_start_time = time.perf_counter()
|
||||||
|
t0 = t6 = time.perf_counter()
|
||||||
|
for i, batch in enumerate(dl, start=1):
|
||||||
|
loop_time = time.perf_counter() - t0
|
||||||
|
t0 = time.perf_counter()
|
||||||
|
dl_time = t0 - t6
|
||||||
|
GlobalCounters.reset()
|
||||||
|
|
||||||
|
mean, logvar = np.split(np.concatenate(batch["npy"], axis=0), 2, axis=1)
|
||||||
|
mean, logvar = Tensor(mean, dtype=dtypes.float32, device="CPU"), Tensor(logvar, dtype=dtypes.float32, device="CPU")
|
||||||
|
tokens = []
|
||||||
|
for text in batch['txt']: tokens += model.cond_stage_model.tokenizer.encode(text, pad_with_zeros=True)
|
||||||
|
tokens = Tensor(tokens, dtype=dtypes.int32, device="CPU").reshape(-1, 77)
|
||||||
|
|
||||||
|
t1 = time.perf_counter()
|
||||||
|
loss, lr = train_step(mean, logvar, tokens, unet, optimizer, lr_scheduler)
|
||||||
|
loss_item, lr_item = loss.item(), lr.item()
|
||||||
|
t2 = time.perf_counter()
|
||||||
|
|
||||||
|
if i == 3:
|
||||||
|
for _ in range(3): ckpt_to_cpu() # do this at the beginning of run to prevent OOM surprises when checkpointing
|
||||||
|
print("BEAM COMPLETE", flush=True) # allows wrapper script to detect BEAM search completion and retry if it failed
|
||||||
|
|
||||||
|
total_train_time = time.perf_counter() - train_start_time
|
||||||
|
if WANDB:
|
||||||
|
wandb.log({"train/loss": loss_item, "train/lr": lr_item, "train/loop_time_prev": loop_time, "train/dl_time": dl_time, "train/step": i,
|
||||||
|
"train/GFLOPS": GlobalCounters.global_ops * 1e-9 / (t2-t1), "train/input_prep_time": t1-t0,
|
||||||
|
"train/train_step_time": t2-t1, "train/total_time": total_train_time})
|
||||||
|
|
||||||
|
if i == 1 and wandb.run is not None:
|
||||||
|
with open(f"{UNET_CKPTDIR}/wandb_run_id_{wandb.run.id}", "w") as f:
|
||||||
|
f.write(f"wandb.run.id = {wandb.run.id}")
|
||||||
|
|
||||||
|
if i % CKPT_STEP_INTERVAL == 0:
|
||||||
|
# https://github.com/mlcommons/training_policies/blob/cfa99da479b8d5931f7a3c67612d021dfb47510a/training_rules.adoc#benchmark_specific_rules
|
||||||
|
# "evaluation is done offline, the time is not counted towards the submission time."
|
||||||
|
fn = f"{UNET_CKPTDIR}/{i}.safetensors"
|
||||||
|
print(f"saving unet checkpoint at {fn}")
|
||||||
|
saved_checkpoints.append(fn)
|
||||||
|
safe_save({k.replace("model.", ""):v for k,v in ckpt_to_cpu().items() if k.startswith("model.")}, fn)
|
||||||
|
if TOTAL_CKPTS and i == TOTAL_CKPTS * CKPT_STEP_INTERVAL:
|
||||||
|
print(f"ending run after {i} steps ({TOTAL_CKPTS} checkpoints collected)")
|
||||||
|
return saved_checkpoints
|
||||||
|
|
||||||
|
t3 = time.perf_counter()
|
||||||
|
print(f"""step {i}: {GlobalCounters.global_ops * 1e-9 / (t2-t1):9.2f} GFLOPS, mem_used: {GlobalCounters.mem_used / 1e9:.2f} GB,
|
||||||
|
loop_time_prev: {loop_time:.2f}, dl_time: {dl_time:.2f}, input_prep_time: {t1-t0:.2f}, train_step_time: {t2-t1:.2f},
|
||||||
|
t3-t2: {t3-t2:.4f}, loss:{loss_item:.5f}, lr:{lr_item:.3e}, total_train_time:{total_train_time:.2f}
|
||||||
|
""")
|
||||||
|
t6 = time.perf_counter()
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
multiprocessing.set_start_method('spawn')
|
multiprocessing.set_start_method('spawn')
|
||||||
|
|
||||||
@@ -1471,7 +1640,7 @@ if __name__ == "__main__":
|
|||||||
else: bench_log_manager = contextlib.nullcontext()
|
else: bench_log_manager = contextlib.nullcontext()
|
||||||
|
|
||||||
with Tensor.train():
|
with Tensor.train():
|
||||||
for m in getenv("MODEL", "resnet,retinanet,unet3d,rnnt,bert,maskrcnn").split(","):
|
for m in getenv("MODEL", "resnet,retinanet,unet3d,rnnt,bert,maskrcnn,stable_diffusion").split(","):
|
||||||
nm = f"train_{m}"
|
nm = f"train_{m}"
|
||||||
if nm in globals():
|
if nm in globals():
|
||||||
print(f"training {m}")
|
print(f"training {m}")
|
||||||
|
|||||||
+57
@@ -0,0 +1,57 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# adapted from https://github.com/mlcommons/training/blob/4bdf5c8ed218ad76565a2ba1ac27c919ccc6d689/stable_diffusion/README.md
|
||||||
|
|
||||||
|
# setup dirs
|
||||||
|
|
||||||
|
DATA=/raid/datasets/stable_diffusion
|
||||||
|
|
||||||
|
LAION=$DATA/laion-400m/webdataset-moments-filtered
|
||||||
|
COCO=$DATA/coco2014
|
||||||
|
mkdir -p $LAION $COCO
|
||||||
|
|
||||||
|
CKPT=/raid/weights/stable_diffusion
|
||||||
|
mkdir -p $CKPT/clip $CKPT/sd $CKPT/inception
|
||||||
|
|
||||||
|
# download data
|
||||||
|
|
||||||
|
# if rclone isn't installed system-wide / in your PATH, put the executable path in quotes below
|
||||||
|
#RCLONE=""
|
||||||
|
RCLONE="rclone"
|
||||||
|
|
||||||
|
## VAE-encoded image latents, from 6.1M image subset of laion-400m
|
||||||
|
## about 1 TB for whole download
|
||||||
|
$RCLONE config create mlc-training s3 provider=Cloudflare access_key_id=76ea42eadb867e854061a1806220ee1e secret_access_key=a53625c4d45e3ca8ac0df8a353ea3a41ffc3292aa25259addd8b7dc5a6ce2936 endpoint=c2686074cb2caf5cbaf6d134bdba8b47.r2.cloudflarestorage.com
|
||||||
|
$RCLONE copy mlc-training:mlcommons-training-wg-public/stable_diffusion/datasets/laion-400m/moments-webdataset-filtered/ ${LAION} --include="*.tar" -P
|
||||||
|
$RCLONE copy mlc-training:mlcommons-training-wg-public/stable_diffusion/datasets/laion-400m/moments-webdataset-filtered/sha512sums.txt ${LAION} -P
|
||||||
|
cd $LAION && grep -E '\.tar$' sha512sums.txt | sha512sum -c --quiet - && \
|
||||||
|
echo "All .tar files verified" || { echo "Checksum failure when validating downloaded Laion moments"; exit 1; }
|
||||||
|
|
||||||
|
## prompts and FID statistics from 30k image subset of coco2014
|
||||||
|
## 33 MB
|
||||||
|
$RCLONE config create mlc-training s3 provider=Cloudflare access_key_id=76ea42eadb867e854061a1806220ee1e secret_access_key=a53625c4d45e3ca8ac0df8a353ea3a41ffc3292aa25259addd8b7dc5a6ce2936 endpoint=c2686074cb2caf5cbaf6d134bdba8b47.r2.cloudflarestorage.com
|
||||||
|
$RCLONE copy mlc-training:mlcommons-training-wg-public/stable_diffusion/datasets/coco2014/val2014_30k.tsv ${COCO} -P
|
||||||
|
|
||||||
|
$RCLONE config create mlc-training s3 provider=Cloudflare access_key_id=76ea42eadb867e854061a1806220ee1e secret_access_key=a53625c4d45e3ca8ac0df8a353ea3a41ffc3292aa25259addd8b7dc5a6ce2936 endpoint=c2686074cb2caf5cbaf6d134bdba8b47.r2.cloudflarestorage.com
|
||||||
|
$RCLONE copy mlc-training:mlcommons-training-wg-public/stable_diffusion/datasets/coco2014/val2014_30k_stats.npz ${COCO} -P
|
||||||
|
|
||||||
|
# download checkpoints
|
||||||
|
|
||||||
|
## clip (needed for text and vision encoders for validation)
|
||||||
|
CLIP_WEIGHTS_URL="https://huggingface.co/laion/CLIP-ViT-H-14-laion2B-s32B-b79K/resolve/main/open_clip_pytorch_model.bin"
|
||||||
|
CLIP_WEIGHTS_SHA256="9a78ef8e8c73fd0df621682e7a8e8eb36c6916cb3c16b291a082ecd52ab79cc4"
|
||||||
|
CLIP_CONFIG_URL="https://huggingface.co/laion/CLIP-ViT-H-14-laion2B-s32B-b79K/raw/main/open_clip_config.json"
|
||||||
|
wget -N -P ${CKPT}/clip ${CLIP_WEIGHTS_URL}
|
||||||
|
wget -N -P ${CKPT}/clip ${CLIP_CONFIG_URL}
|
||||||
|
echo "${CLIP_WEIGHTS_SHA256} ${CKPT}/clip/open_clip_pytorch_model.bin" | sha256sum -c
|
||||||
|
|
||||||
|
## sd (needed for latent->image decoder for validation, also has clip text encoder for training)
|
||||||
|
SD_WEIGHTS_URL='https://huggingface.co/stabilityai/stable-diffusion-2-base/resolve/main/512-base-ema.ckpt'
|
||||||
|
SD_WEIGHTS_SHA256="d635794c1fedfdfa261e065370bea59c651fc9bfa65dc6d67ad29e11869a1824"
|
||||||
|
wget -N -P ${CKPT}/sd ${SD_WEIGHTS_URL}
|
||||||
|
echo "${SD_WEIGHTS_SHA256} ${CKPT}/sd/512-base-ema.ckpt" | sha256sum -c
|
||||||
|
|
||||||
|
## inception (needed for validation)
|
||||||
|
FID_WEIGHTS_URL='https://github.com/mseitzer/pytorch-fid/releases/download/fid_weights/pt_inception-2015-12-05-6726825d.pth'
|
||||||
|
FID_WEIGHTS_SHA1="bd836944fd6db519dfd8d924aa457f5b3c8357ff"
|
||||||
|
wget -N -P ${CKPT}/inception ${FID_WEIGHTS_URL}
|
||||||
|
echo "${FID_WEIGHTS_SHA1} ${CKPT}/inception/pt_inception-2015-12-05-6726825d.pth" | sha1sum -c
|
||||||
+72
@@ -0,0 +1,72 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
|
||||||
|
DATETIME=${2:-$(date "+%m%d%H%M")}
|
||||||
|
LOGFILE="${HOME}/logs/sd_mi300x_${DATETIME}.log"
|
||||||
|
# UNET_CKPTDIR must be set: training saves checkpoints to this path, then a separate eval process scans this path to know which checkpoints to eval
|
||||||
|
export UNET_CKPTDIR="${HOME}/stable_diffusion/training_checkpoints/${DATETIME}"
|
||||||
|
mkdir -p "${HOME}/logs" "$UNET_CKPTDIR"
|
||||||
|
|
||||||
|
# run this script in isolation when using the --bg flag
|
||||||
|
if [[ "${1:-}" == "--bg" ]]; then
|
||||||
|
echo "logging output to $LOGFILE"
|
||||||
|
echo "saving UNet checkpoints to $UNET_CKPTDIR"
|
||||||
|
script_path="$(readlink -f "${BASH_SOURCE[0]}")"
|
||||||
|
nohup bash "$script_path" run "$DATETIME" >"$LOGFILE" 2>&1 & disown $!
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# venv management
|
||||||
|
if [[ -d .venv-sd-mlperf ]]; then
|
||||||
|
. .venv-sd-mlperf/bin/activate
|
||||||
|
else
|
||||||
|
python3 -m venv .venv-sd-mlperf && . .venv-sd-mlperf/bin/activate
|
||||||
|
pip install --index-url https://download.pytorch.org/whl/cpu torch && pip install tqdm numpy ftfy regex pillow scipy wandb webdataset
|
||||||
|
fi
|
||||||
|
pip list
|
||||||
|
apt list --installed | grep amdgpu
|
||||||
|
rocm-smi --version
|
||||||
|
modinfo amdgpu | grep version
|
||||||
|
|
||||||
|
export BEAM=2 BEAM_UOPS_MAX=8000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 IGNORE_JIT_FIRST_BEAM=1 HCQDEV_WAIT_TIMEOUT_MS=300000
|
||||||
|
export AMD_LLVM=0 # bf16 seems to require this
|
||||||
|
export DATADIR="/raid/datasets/stable_diffusion"
|
||||||
|
export CKPTDIR="/raid/weights/stable_diffusion"
|
||||||
|
export EVAL_CKPT_DIR=$UNET_CKPTDIR
|
||||||
|
export MODEL="stable_diffusion" PYTHONPATH="."
|
||||||
|
export GPUS=8 BS=304
|
||||||
|
export CONTEXT_BS=816 DENOISE_BS=600 DECODE_BS=384 INCEPTION_BS=560 CLIP_BS=240
|
||||||
|
export WANDB=1
|
||||||
|
export PARALLEL=4
|
||||||
|
export PYTHONUNBUFFERED=1
|
||||||
|
sudo rocm-smi -d 0 1 2 3 4 5 6 7 --setperfdeterminism 1500 || exit 1
|
||||||
|
|
||||||
|
# Retry BEAM search if script fails before BEAM COMPLETE is printed, but don't retry after that
|
||||||
|
run_retry(){ local try=0 max=5 code tmp py pgid kids
|
||||||
|
while :; do
|
||||||
|
tmp=$(mktemp)
|
||||||
|
setsid bash -c 'exec env "$@"' _ "$@" > >(tee -a "$LOGFILE" | tee "$tmp") 2>&1 &
|
||||||
|
py=$!; pgid=$(ps -o pgid= -p "$py" | tr -d ' ')
|
||||||
|
wait "$py"; code=$?
|
||||||
|
[[ -n "$pgid" ]] && { kill -TERM -"$pgid" 2>/dev/null; sleep 1; kill -KILL -"$pgid" 2>/dev/null; }
|
||||||
|
kids=$(pgrep -P "$py" || true)
|
||||||
|
while [[ -n "$kids" ]]; do
|
||||||
|
kill -TERM $kids 2>/dev/null; sleep 0.5
|
||||||
|
kids=$(for k in $kids; do pgrep -P "$k" || true; done)
|
||||||
|
done
|
||||||
|
grep -q 'BEAM COMPLETE' "$tmp" && { rm -f "$tmp"; return 1; }
|
||||||
|
rm -f "$tmp"
|
||||||
|
((code==0)) && return 0
|
||||||
|
((try>=max)) && return 2
|
||||||
|
((try++)); sleep 90; echo "try = ${try}"
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
|
# Power limiting to 400W is only needed if GPUs fall out of sync (causing 2.2x increased train time) at higher power, which has been observed at 450W
|
||||||
|
sudo rocm-smi -d 0 1 2 3 4 5 6 7 --setpoweroverdrive 750 && \
|
||||||
|
run_retry TOTAL_CKPTS=7 python3 examples/mlperf/model_train.py; (( $? == 2 )) && { echo "training failed before BEAM completion"; exit 2; }
|
||||||
|
sleep 90
|
||||||
|
|
||||||
|
run_retry EVAL_SAMPLES=600 python3 examples/mlperf/model_eval.py; (( $? == 2 )) && { echo "eval failed before BEAM completion"; exit 2; }
|
||||||
|
# Checkpoints will be evaluated in reverse chronological order, even if above training crashed early
|
||||||
|
# STOP_IF_CONVERGED=1: Stop the eval after the first time convergence is detected; no more checkpoints will be evaluated after that.
|
||||||
|
STOP_IF_CONVERGED=1 python3 examples/mlperf/model_eval.py
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=1 BS=128 EVAL_BS=128
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
|
||||||
|
export BEAM=3 BEAM_UOPS_MAX=4000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
# export BEAM_LOG_SURPASS_MAX=1
|
||||||
|
# export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export RESET_STEP=1
|
||||||
|
export BENCHMARK=10 BERT_LAYERS=2 DEBUG=2
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+69
@@ -0,0 +1,69 @@
|
|||||||
|
# 1. Problem
|
||||||
|
|
||||||
|
This problem uses BERT for NLP.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
Install tinygrad and mlperf-logging (uncomment mlperf from setup.py) from branch mlperf_training_v5.0.
|
||||||
|
```
|
||||||
|
git clone https://github.com/tinygrad/tinygrad.git
|
||||||
|
python3 -m pip install -e ".[mlperf]"
|
||||||
|
```
|
||||||
|
Also install gdown (for dataset), numpy, tqdm and tensorflow.
|
||||||
|
```
|
||||||
|
pip install gdown numpy tqdm tensorflow
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
Install the p2p driver per [README](https://github.com/tinygrad/open-gpu-kernel-modules/blob/550.54.15-p2p/README.md)
|
||||||
|
This is the default on production tinybox green.
|
||||||
|
|
||||||
|
# 2. Directions
|
||||||
|
|
||||||
|
## Steps to download and verify data
|
||||||
|
|
||||||
|
### 1. Download raw data
|
||||||
|
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" WIKI_TRAIN=1 VERIFY_CHECKSUM=1 python3 extra/datasets/wikipedia_download.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Preprocess train and validation data
|
||||||
|
|
||||||
|
Note: The number of threads used for preprocessing is limited by available memory. With 128GB of RAM, a maximum of 16 threads is recommended.
|
||||||
|
|
||||||
|
#### Training:
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" NUM_WORKERS=16 python3 extra/datasets/wikipedia.py pre-train all
|
||||||
|
```
|
||||||
|
|
||||||
|
Generating a specific topic (Between 0 and 499)
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" python3 extra/datasets/wikipedia.py pre-train 42
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Validation:
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" python3 extra/datasets/wikipedia.py pre-eval
|
||||||
|
```
|
||||||
|
## Running
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_green/run_and_time.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_red/run_and_time.sh
|
||||||
|
```
|
||||||
|
### tinybox_8xMI300X
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_8xMI300X/run_and_time.sh
|
||||||
|
```
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=8 BS=1024 EVAL_BS=1024
|
||||||
|
export OPT_BASE_LEARNING_RATE=0.0011 OPT_LAMB_BETA_1=0.60466 OPT_LAMB_BETA_2=0.85437 DECAY=0.1
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=3 BEAM_UOPS_MAX=6000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1 FREE_INTERMEDIATE=0
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export BENCHMARK=10 BERT_LAYERS=2
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+20
@@ -0,0 +1,20 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=8 BS=1024 EVAL_BS=1024
|
||||||
|
|
||||||
|
# similar to https://github.com/mlcommons/training_results_v3.1/blob/d06288b2bd675a9d88e0e6181f5bb5626b71ec19/Quanta_Cloud_Technology/results/D54U-3U/bert/result_1.txt#L54
|
||||||
|
export OPT_BASE_LEARNING_RATE=0.0011 OPT_LAMB_BETA_1=0.60466 OPT_LAMB_BETA_2=0.85437 DECAY=0.1
|
||||||
|
export TRAIN_STEPS=3900
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=3 BEAM_UOPS_MAX=6000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1 FREE_INTERMEDIATE=0
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
|
||||||
|
RUNMLPERF=1 python3 examples/mlperf/model_train.py
|
||||||
+31
@@ -0,0 +1,31 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -e # Exit on any error
|
||||||
|
set -o pipefail # Make pipeline fail if any command fails
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export SUBMISSION_PLATFORM="tinybox_8xMI300X"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=8 BS=1024 EVAL_BS=1024
|
||||||
|
|
||||||
|
# similar to https://github.com/mlcommons/training_results_v3.1/blob/d06288b2bd675a9d88e0e6181f5bb5626b71ec19/Quanta_Cloud_Technology/results/D54U-3U/bert/result_1.txt#L54
|
||||||
|
export OPT_BASE_LEARNING_RATE=0.0011 OPT_LAMB_BETA_1=0.60466 OPT_LAMB_BETA_2=0.85437 DECAY=0.1
|
||||||
|
export TRAIN_STEPS=3900
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=3 BEAM_UOPS_MAX=6000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1 FREE_INTERMEDIATE=0
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
# pip install -e ".[mlperf]"
|
||||||
|
export LOGMLPERF=1
|
||||||
|
|
||||||
|
export SEED=$RANDOM
|
||||||
|
DATETIME=$(date "+%m%d%H%M")
|
||||||
|
LOGFILE="bert_8xMI300x_${DATETIME}_${SEED}.log"
|
||||||
|
|
||||||
|
BENCHMARK=10 INITMLPERF=1 BERT_LAYERS=2 python3 examples/mlperf/model_train.py | tee $LOGFILE
|
||||||
|
|
||||||
|
# run
|
||||||
|
PARALLEL=0 RUNMLPERF=1 python3 examples/mlperf/model_train.py | tee -a $LOGFILE
|
||||||
+69
@@ -0,0 +1,69 @@
|
|||||||
|
# 1. Problem
|
||||||
|
|
||||||
|
This problem uses BERT for NLP.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
Install tinygrad and mlperf-logging (uncomment mlperf from setup.py) from branch mlperf_training_v5.0.
|
||||||
|
```
|
||||||
|
git clone https://github.com/tinygrad/tinygrad.git
|
||||||
|
python3 -m pip install -e ".[mlperf]"
|
||||||
|
```
|
||||||
|
Also install gdown (for dataset), numpy, tqdm and tensorflow.
|
||||||
|
```
|
||||||
|
pip install gdown numpy tqdm tensorflow
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
Install the p2p driver per [README](https://github.com/tinygrad/open-gpu-kernel-modules/blob/550.54.15-p2p/README.md)
|
||||||
|
This is the default on production tinybox green.
|
||||||
|
|
||||||
|
# 2. Directions
|
||||||
|
|
||||||
|
## Steps to download and verify data
|
||||||
|
|
||||||
|
### 1. Download raw data
|
||||||
|
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" WIKI_TRAIN=1 VERIFY_CHECKSUM=1 python3 extra/datasets/wikipedia_download.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Preprocess train and validation data
|
||||||
|
|
||||||
|
Note: The number of threads used for preprocessing is limited by available memory. With 128GB of RAM, a maximum of 16 threads is recommended.
|
||||||
|
|
||||||
|
#### Training:
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" NUM_WORKERS=16 python3 extra/datasets/wikipedia.py pre-train all
|
||||||
|
```
|
||||||
|
|
||||||
|
Generating a specific topic (Between 0 and 499)
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" python3 extra/datasets/wikipedia.py pre-train 42
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Validation:
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" python3 extra/datasets/wikipedia.py pre-eval
|
||||||
|
```
|
||||||
|
## Running
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_green/run_and_time.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_red/run_and_time.sh
|
||||||
|
```
|
||||||
|
### tinybox_8xMI300X
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_8xMI300X/run_and_time.sh
|
||||||
|
```
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" SUM_DTYPE="HALF" GPUS=6 BS=90 EVAL_BS=90
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=8 BEAM_UOPS_MAX=10000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BEAM_LOG_SURPASS_MAX=1
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export BENCHMARK=10 BERT_LAYERS=2 DEBUG=2
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" SUM_DTYPE="HALF" GPUS=6 BS=90 EVAL_BS=90
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=8 BEAM_UOPS_MAX=10000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
|
||||||
|
RUNMLPERF=1 python3 examples/mlperf/model_train.py
|
||||||
+28
@@ -0,0 +1,28 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -e # Exit on any error
|
||||||
|
set -o pipefail # Make pipeline fail if any command fails
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export SUBMISSION_PLATFORM="tinybox_green"
|
||||||
|
export DEFAULT_FLOAT="HALF" SUM_DTYPE="HALF" GPUS=6 BS=90 EVAL_BS=90
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=8 BEAM_UOPS_MAX=10000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
# pip install -e ".[mlperf]"
|
||||||
|
export LOGMLPERF=1
|
||||||
|
|
||||||
|
export SEED=$RANDOM
|
||||||
|
DATETIME=$(date "+%m%d%H%M")
|
||||||
|
LOGFILE="bert_green_${DATETIME}_${SEED}.log"
|
||||||
|
|
||||||
|
# init
|
||||||
|
BENCHMARK=10 INITMLPERF=1 BERT_LAYERS=2 python3 examples/mlperf/model_train.py | tee $LOGFILE
|
||||||
|
|
||||||
|
# run
|
||||||
|
PARALLEL=0 RUNMLPERF=1 python3 examples/mlperf/model_train.py | tee -a $LOGFILE
|
||||||
+69
@@ -0,0 +1,69 @@
|
|||||||
|
# 1. Problem
|
||||||
|
|
||||||
|
This problem uses BERT for NLP.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
Install tinygrad and mlperf-logging (uncomment mlperf from setup.py) from branch mlperf_training_v5.0.
|
||||||
|
```
|
||||||
|
git clone https://github.com/tinygrad/tinygrad.git
|
||||||
|
python3 -m pip install -e ".[mlperf]"
|
||||||
|
```
|
||||||
|
Also install gdown (for dataset), numpy, tqdm and tensorflow.
|
||||||
|
```
|
||||||
|
pip install gdown numpy tqdm tensorflow
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
Install the p2p driver per [README](https://github.com/tinygrad/open-gpu-kernel-modules/blob/550.54.15-p2p/README.md)
|
||||||
|
This is the default on production tinybox green.
|
||||||
|
|
||||||
|
# 2. Directions
|
||||||
|
|
||||||
|
## Steps to download and verify data
|
||||||
|
|
||||||
|
### 1. Download raw data
|
||||||
|
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" WIKI_TRAIN=1 VERIFY_CHECKSUM=1 python3 extra/datasets/wikipedia_download.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Preprocess train and validation data
|
||||||
|
|
||||||
|
Note: The number of threads used for preprocessing is limited by available memory. With 128GB of RAM, a maximum of 16 threads is recommended.
|
||||||
|
|
||||||
|
#### Training:
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" NUM_WORKERS=16 python3 extra/datasets/wikipedia.py pre-train all
|
||||||
|
```
|
||||||
|
|
||||||
|
Generating a specific topic (Between 0 and 499)
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" python3 extra/datasets/wikipedia.py pre-train 42
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Validation:
|
||||||
|
```
|
||||||
|
BASEDIR="/raid/datasets/wiki" python3 extra/datasets/wikipedia.py pre-eval
|
||||||
|
```
|
||||||
|
## Running
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_green/run_and_time.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_red/run_and_time.sh
|
||||||
|
```
|
||||||
|
### tinybox_8xMI300X
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/bert/implementations/tinybox_8xMI300X/run_and_time.sh
|
||||||
|
```
|
||||||
+18
@@ -0,0 +1,18 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" SUM_DTYPE="HALF" GPUS=6 BS=90 EVAL_BS=90
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=5 BEAM_UOPS_MAX=8000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BEAM_LOG_SURPASS_MAX=1
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export RESET_STEP=1
|
||||||
|
export BENCHMARK=10 BERT_LAYERS=2 DEBUG=2
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+16
@@ -0,0 +1,16 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export DEFAULT_FLOAT="HALF" SUM_DTYPE="HALF" GPUS=6 BS=90 EVAL_BS=90
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=5 BEAM_UOPS_MAX=8000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
|
||||||
|
RUNMLPERF=1 python3 examples/mlperf/model_train.py
|
||||||
+31
@@ -0,0 +1,31 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -e # Exit on any error
|
||||||
|
set -o pipefail # Make pipeline fail if any command fails
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="bert"
|
||||||
|
export SUBMISSION_PLATFORM="tinybox_red"
|
||||||
|
export DEFAULT_FLOAT="HALF" SUM_DTYPE="HALF" GPUS=6 BS=90 EVAL_BS=90
|
||||||
|
|
||||||
|
export IGNORE_OOB=1
|
||||||
|
export REWRITE_STACK_LIMIT=500000
|
||||||
|
|
||||||
|
export BEAM=5 BEAM_UOPS_MAX=8000 BEAM_UPCAST_MAX=256 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BASEDIR="/raid/datasets/wiki"
|
||||||
|
|
||||||
|
# pip install -e ".[mlperf]"
|
||||||
|
export LOGMLPERF=1
|
||||||
|
|
||||||
|
export SEED=$RANDOM
|
||||||
|
DATETIME=$(date "+%m%d%H%M")
|
||||||
|
LOGFILE="bert_red_${DATETIME}_${SEED}.log"
|
||||||
|
|
||||||
|
export HCQDEV_WAIT_TIMEOUT_MS=100000 # prevents hang?
|
||||||
|
|
||||||
|
# init
|
||||||
|
sleep 5 && sudo rmmod amdgpu || true
|
||||||
|
BENCHMARK=10 INITMLPERF=1 BERT_LAYERS=2 python3 examples/mlperf/model_train.py | tee $LOGFILE
|
||||||
|
|
||||||
|
# run
|
||||||
|
PARALLEL=0 RUNMLPERF=1 python3 examples/mlperf/model_train.py | tee -a $LOGFILE
|
||||||
+50
@@ -0,0 +1,50 @@
|
|||||||
|
# 1. Problem
|
||||||
|
|
||||||
|
This problem uses the ResNet-50 CNN to do image classification.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
Install tinygrad and mlperf-logging from master.
|
||||||
|
```
|
||||||
|
git clone https://github.com/tinygrad/tinygrad.git
|
||||||
|
python3 -m pip install -e ".[mlperf]"
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
Install the p2p driver per [README](https://github.com/tinygrad/open-gpu-kernel-modules/blob/550.54.15-p2p/README.md)
|
||||||
|
This is the default on production tinybox green.
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
Disable cwsr
|
||||||
|
This is the default on production tinybox red.
|
||||||
|
```
|
||||||
|
sudo vi /etc/modprobe.d/amdgpu.conf
|
||||||
|
cat <<EOF > /etc/modprobe.d/amdgpu.conf
|
||||||
|
options amdgpu cwsr_enable=0
|
||||||
|
EOF
|
||||||
|
sudo update-initramfs -u
|
||||||
|
sudo reboot
|
||||||
|
|
||||||
|
# validate
|
||||||
|
sudo cat /sys/module/amdgpu/parameters/cwsr_enable #= 0
|
||||||
|
```
|
||||||
|
|
||||||
|
# 2. Directions
|
||||||
|
|
||||||
|
## Steps to download and verify data
|
||||||
|
|
||||||
|
```
|
||||||
|
IMGNET_TRAIN=1 python3 extra/datasets/imagenet_download.py
|
||||||
|
```
|
||||||
|
|
||||||
|
## Steps for one time setup
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v4.0/tinycorp/benchmarks/resnet/implementations/tinybox_red/setup.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
## Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v4.0/tinycorp/benchmarks/resnet/implementations/tinybox_red/run_and_time.sh
|
||||||
|
```
|
||||||
+13
@@ -0,0 +1,13 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="resnet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=1536 EVAL_BS=192
|
||||||
|
|
||||||
|
export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=4 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=10 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export BENCHMARK=10 DEBUG=2
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="resnet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=1536 EVAL_BS=192
|
||||||
|
|
||||||
|
export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=4 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=10 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export EVAL_START_EPOCH=3 EVAL_FREQ=4
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+25
@@ -0,0 +1,25 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -e # Exit on any error
|
||||||
|
set -o pipefail # Make pipeline fail if any command fails
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="resnet"
|
||||||
|
export SUBMISSION_PLATFORM="tinybox_green"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=1536 EVAL_BS=192
|
||||||
|
|
||||||
|
export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=4 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=10 BEAM_PADTO=0
|
||||||
|
|
||||||
|
# pip install -e ".[mlperf]"
|
||||||
|
export LOGMLPERF=${LOGMLPERF:-1}
|
||||||
|
|
||||||
|
export SEED=$RANDOM
|
||||||
|
DATETIME=$(date "+%m%d%H%M")
|
||||||
|
LOGFILE="resnet_green_${DATETIME}_${SEED}.log"
|
||||||
|
|
||||||
|
# init
|
||||||
|
BENCHMARK=10 INITMLPERF=1 python3 examples/mlperf/model_train.py | tee $LOGFILE
|
||||||
|
|
||||||
|
# run
|
||||||
|
PARALLEL=0 RUNMLPERF=1 EVAL_START_EPOCH=3 EVAL_FREQ=4 python3 examples/mlperf/model_train.py | tee -a $LOGFILE
|
||||||
+50
@@ -0,0 +1,50 @@
|
|||||||
|
# 1. Problem
|
||||||
|
|
||||||
|
This problem uses the ResNet-50 CNN to do image classification.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
Install tinygrad and mlperf-logging from master.
|
||||||
|
```
|
||||||
|
git clone https://github.com/tinygrad/tinygrad.git
|
||||||
|
python3 -m pip install -e ".[mlperf]"
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
Install the p2p driver per [README](https://github.com/tinygrad/open-gpu-kernel-modules/blob/550.54.15-p2p/README.md)
|
||||||
|
This is the default on production tinybox green.
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
Disable cwsr
|
||||||
|
This is the default on production tinybox red.
|
||||||
|
```
|
||||||
|
sudo vi /etc/modprobe.d/amdgpu.conf
|
||||||
|
cat <<EOF > /etc/modprobe.d/amdgpu.conf
|
||||||
|
options amdgpu cwsr_enable=0
|
||||||
|
EOF
|
||||||
|
sudo update-initramfs -u
|
||||||
|
sudo reboot
|
||||||
|
|
||||||
|
# validate
|
||||||
|
sudo cat /sys/module/amdgpu/parameters/cwsr_enable #= 0
|
||||||
|
```
|
||||||
|
|
||||||
|
# 2. Directions
|
||||||
|
|
||||||
|
## Steps to download and verify data
|
||||||
|
|
||||||
|
```
|
||||||
|
IMGNET_TRAIN=1 python3 extra/datasets/imagenet_download.py
|
||||||
|
```
|
||||||
|
|
||||||
|
## Steps for one time setup
|
||||||
|
|
||||||
|
### tinybox_red
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v4.0/tinycorp/benchmarks/resnet/implementations/tinybox_red/setup.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
## Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v4.0/tinycorp/benchmarks/resnet/implementations/tinybox_red/run_and_time.sh
|
||||||
|
```
|
||||||
+13
@@ -0,0 +1,13 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="resnet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=1536 EVAL_BS=192
|
||||||
|
|
||||||
|
export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=4 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=2000 BEAM_UPCAST_MAX=96 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export BENCHMARK=10 DEBUG=${DEBUG:-2}
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="resnet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=1536 EVAL_BS=192
|
||||||
|
|
||||||
|
export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=4 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=2000 BEAM_UPCAST_MAX=96 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export EVAL_START_EPOCH=3 EVAL_FREQ=4
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
|
||||||
|
python3 examples/mlperf/model_train.py
|
||||||
+26
@@ -0,0 +1,26 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -e # Exit on any error
|
||||||
|
set -o pipefail # Make pipeline fail if any command fails
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="resnet"
|
||||||
|
export SUBMISSION_PLATFORM="tinybox_red"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=1536 EVAL_BS=192
|
||||||
|
|
||||||
|
export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=4 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=2000 BEAM_UPCAST_MAX=96 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
# pip install -e ".[mlperf]"
|
||||||
|
export LOGMLPERF=${LOGMLPERF:-1}
|
||||||
|
|
||||||
|
export SEED=$RANDOM
|
||||||
|
DATETIME=$(date "+%m%d%H%M")
|
||||||
|
LOGFILE="resnet_red_${DATETIME}_${SEED}.log"
|
||||||
|
|
||||||
|
# init
|
||||||
|
sleep 5 && sudo rmmod amdgpu || true
|
||||||
|
BENCHMARK=10 INITMLPERF=1 python3 examples/mlperf/model_train.py | tee $LOGFILE
|
||||||
|
|
||||||
|
# run
|
||||||
|
PARALLEL=0 RUNMLPERF=1 EVAL_START_EPOCH=3 EVAL_FREQ=4 python3 examples/mlperf/model_train.py | tee -a $LOGFILE
|
||||||
+8
@@ -0,0 +1,8 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
rocm-smi --setprofile compute
|
||||||
|
rocm-smi --setmclk 3
|
||||||
|
rocm-smi --setperflevel high
|
||||||
|
|
||||||
|
# power cap to 350W
|
||||||
|
echo "350000000" | sudo tee /sys/class/drm/card{1..6}/device/hwmon/hwmon*/power1_cap
|
||||||
+38
@@ -0,0 +1,38 @@
|
|||||||
|
# 1. Problem
|
||||||
|
|
||||||
|
This problem uses RetinaNet for SSD.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
Install tinygrad and mlperf-logging (uncomment mlperf from setup.py) from branch mlperf_training_v5.0.
|
||||||
|
```
|
||||||
|
git clone https://github.com/tinygrad/tinygrad.git
|
||||||
|
python3 -m pip install -e ".[mlperf]"
|
||||||
|
```
|
||||||
|
|
||||||
|
Also install the following dependencies:
|
||||||
|
```
|
||||||
|
pip install tqdm numpy pycocotools boto3 pandas torch torchvision
|
||||||
|
```
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
Install the p2p driver per [README](https://github.com/tinygrad/open-gpu-kernel-modules/blob/550.54.15-p2p/README.md)
|
||||||
|
This is the default on production tinybox green.
|
||||||
|
|
||||||
|
# 2. Directions
|
||||||
|
|
||||||
|
## Steps to download data
|
||||||
|
|
||||||
|
Run the following:
|
||||||
|
```
|
||||||
|
BASEDIR=/raid/datasets/openimages python3 extra/datasets/openimages.py
|
||||||
|
```
|
||||||
|
|
||||||
|
## Running
|
||||||
|
|
||||||
|
### tinybox_green
|
||||||
|
|
||||||
|
#### Steps to run benchmark
|
||||||
|
```
|
||||||
|
examples/mlperf/training_submission_v5.0/tinycorp/benchmarks/retinanet/implementations/tinybox_green/run_and_time.sh
|
||||||
|
```
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="retinanet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=96 EVAL_BS=96
|
||||||
|
export BASEDIR="/raid/datasets/openimages"
|
||||||
|
|
||||||
|
# export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=2 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export BENCHMARK=5 DEBUG=2
|
||||||
|
|
||||||
|
python examples/mlperf/model_train.py
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="retinanet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=96 EVAL_BS=96
|
||||||
|
export BASEDIR="/raid/datasets/openimages"
|
||||||
|
|
||||||
|
# export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=2 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
export RUNMLPERF=1
|
||||||
|
|
||||||
|
python examples/mlperf/model_train.py
|
||||||
+25
@@ -0,0 +1,25 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
set -e # Exit on any error
|
||||||
|
set -o pipefail # Make pipeline fail if any command fails
|
||||||
|
|
||||||
|
export PYTHONPATH="." NV=1
|
||||||
|
export MODEL="retinanet"
|
||||||
|
export SUBMISSION_PLATFORM="tinybox_green"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=96 EVAL_BS=96
|
||||||
|
|
||||||
|
export TRAIN_BEAM=2 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
export IGNORE_JIT_FIRST_BEAM=1
|
||||||
|
export BASEDIR="/raid/datasets/openimages"
|
||||||
|
|
||||||
|
# pip install -e ".[mlperf]"
|
||||||
|
export LOGMLPERF=1
|
||||||
|
|
||||||
|
export SEED=$RANDOM
|
||||||
|
DATETIME=$(date "+%m%d%H%M")
|
||||||
|
LOGFILE="retinanet_green_${DATETIME}_${SEED}.log"
|
||||||
|
|
||||||
|
# init
|
||||||
|
BENCHMARK=10 INITMLPERF=1 python3 examples/mlperf/model_train.py | tee $LOGFILE
|
||||||
|
|
||||||
|
# run
|
||||||
|
PARALLEL=0 RUNMLPERF=1 python3 examples/mlperf/model_train.py | tee -a $LOGFILE
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="retinanet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=96 EVAL_BS=96
|
||||||
|
export BASEDIR="/raid/datasets/openimages"
|
||||||
|
|
||||||
|
# export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=2 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export BENCHMARK=5 DEBUG=2
|
||||||
|
|
||||||
|
python examples/mlperf/model_train.py
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
export PYTHONPATH="." AMD=1
|
||||||
|
export MODEL="retinanet"
|
||||||
|
export DEFAULT_FLOAT="HALF" GPUS=6 BS=96 EVAL_BS=96
|
||||||
|
export BASEDIR="/raid/datasets/openimages"
|
||||||
|
|
||||||
|
# export RESET_STEP=0
|
||||||
|
|
||||||
|
export TRAIN_BEAM=2 IGNORE_JIT_FIRST_BEAM=1 BEAM_UOPS_MAX=1500 BEAM_UPCAST_MAX=64 BEAM_LOCAL_MAX=1024 BEAM_MIN_PROGRESS=5 BEAM_PADTO=0
|
||||||
|
|
||||||
|
export WANDB=1 PARALLEL=0
|
||||||
|
export RUNMLPERF=1
|
||||||
|
|
||||||
|
python examples/mlperf/model_train.py
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
{
|
||||||
|
"submitter": "tinycorp",
|
||||||
|
"division": "closed",
|
||||||
|
"status": "Available on-premise",
|
||||||
|
"system_name": "tinybox 8xMI300X",
|
||||||
|
"number_of_nodes": "1",
|
||||||
|
"host_processors_per_node": "2",
|
||||||
|
"host_processor_model_name": "AMD EPYC 9354",
|
||||||
|
"host_processor_core_count": "32",
|
||||||
|
"host_processor_vcpu_count": "64",
|
||||||
|
"host_processor_frequency": "",
|
||||||
|
"host_processor_caches": "",
|
||||||
|
"host_processor_interconnect": "",
|
||||||
|
"host_memory_capacity": "2304GB",
|
||||||
|
"host_storage_type": "NVMe SSD",
|
||||||
|
"host_storage_capacity": "3x 4TB raid array",
|
||||||
|
"host_networking": "",
|
||||||
|
"host_networking_topology": "",
|
||||||
|
"host_memory_configuration": "24x 96GB DDR5",
|
||||||
|
"accelerators_per_node": "8",
|
||||||
|
"accelerator_model_name": "AMD Instinct MI300X 192GB HBM3",
|
||||||
|
"accelerator_host_interconnect": "PCIe 5.0 x16",
|
||||||
|
"accelerator_frequency": "",
|
||||||
|
"accelerator_on-chip_memories": "",
|
||||||
|
"accelerator_memory_configuration": "HBM3",
|
||||||
|
"accelerator_memory_capacity": "192GB",
|
||||||
|
"accelerator_interconnect": "",
|
||||||
|
"accelerator_interconnect_topology": "",
|
||||||
|
"cooling": "air",
|
||||||
|
"hw_notes": "",
|
||||||
|
"framework": "tinygrad, branch mlperf_training_v5.0",
|
||||||
|
"other_software_stack": {
|
||||||
|
"python": "3.10.16",
|
||||||
|
"ROCm": "3.0.0+94441cb"
|
||||||
|
},
|
||||||
|
"operating_system": "Ubuntu 24.04.1 LTS",
|
||||||
|
"sw_notes": ""
|
||||||
|
}
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
{
|
||||||
|
"submitter": "tinycorp",
|
||||||
|
"division": "closed",
|
||||||
|
"status": "Available on-premise",
|
||||||
|
"system_name": "tinybox green",
|
||||||
|
"number_of_nodes": "1",
|
||||||
|
"host_processors_per_node": "1",
|
||||||
|
"host_processor_model_name": "AMD EPYC 7532",
|
||||||
|
"host_processor_core_count": "32",
|
||||||
|
"host_processor_vcpu_count": "64",
|
||||||
|
"host_processor_frequency": "",
|
||||||
|
"host_processor_caches": "",
|
||||||
|
"host_processor_interconnect": "",
|
||||||
|
"host_memory_capacity": "128GB",
|
||||||
|
"host_storage_type": "NVMe SSD",
|
||||||
|
"host_storage_capacity": "4 TB raid array + 1 TB boot",
|
||||||
|
"host_networking": "",
|
||||||
|
"host_networking_topology": "",
|
||||||
|
"host_memory_configuration": "8x 16GB DDR4",
|
||||||
|
"accelerators_per_node": "6",
|
||||||
|
"accelerator_model_name": "NVIDIA GeForce RTX 4090",
|
||||||
|
"accelerator_host_interconnect": "PCIe 4.0 x16",
|
||||||
|
"accelerator_frequency": "",
|
||||||
|
"accelerator_on-chip_memories": "",
|
||||||
|
"accelerator_memory_configuration": "GDDR6X",
|
||||||
|
"accelerator_memory_capacity": "24GB",
|
||||||
|
"accelerator_interconnect": "",
|
||||||
|
"accelerator_interconnect_topology": "",
|
||||||
|
"cooling": "air",
|
||||||
|
"hw_notes": "",
|
||||||
|
"framework": "tinygrad, branch mlperf_training_v5.0",
|
||||||
|
"other_software_stack": {
|
||||||
|
"python": "3.10.12",
|
||||||
|
"CUDA": "12.4"
|
||||||
|
},
|
||||||
|
"operating_system": "Ubuntu 22.04.4",
|
||||||
|
"sw_notes": ""
|
||||||
|
}
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
{
|
||||||
|
"submitter": "tinycorp",
|
||||||
|
"division": "closed",
|
||||||
|
"status": "Available on-premise",
|
||||||
|
"system_name": "tinybox red",
|
||||||
|
"number_of_nodes": "1",
|
||||||
|
"host_processors_per_node": "1",
|
||||||
|
"host_processor_model_name": "AMD EPYC 7532",
|
||||||
|
"host_processor_core_count": "32",
|
||||||
|
"host_processor_vcpu_count": "64",
|
||||||
|
"host_processor_frequency": "",
|
||||||
|
"host_processor_caches": "",
|
||||||
|
"host_processor_interconnect": "",
|
||||||
|
"host_memory_capacity": "128GB",
|
||||||
|
"host_storage_type": "NVMe SSD",
|
||||||
|
"host_storage_capacity": "4 TB raid array + 1 TB boot",
|
||||||
|
"host_networking": "",
|
||||||
|
"host_networking_topology": "",
|
||||||
|
"host_memory_configuration": "8x 16GB DDR4",
|
||||||
|
"accelerators_per_node": "6",
|
||||||
|
"accelerator_model_name": "AMD Radeon RX 7900 XTX",
|
||||||
|
"accelerator_host_interconnect": "PCIe 4.0 x16",
|
||||||
|
"accelerator_frequency": "",
|
||||||
|
"accelerator_on-chip_memories": "",
|
||||||
|
"accelerator_memory_configuration": "GDDR6",
|
||||||
|
"accelerator_memory_capacity": "24GB",
|
||||||
|
"accelerator_interconnect": "",
|
||||||
|
"accelerator_interconnect_topology": "",
|
||||||
|
"cooling": "air",
|
||||||
|
"hw_notes": "",
|
||||||
|
"framework": "tinygrad, branch mlperf_training_v5.0",
|
||||||
|
"other_software_stack": {
|
||||||
|
"python": "3.10.12"
|
||||||
|
},
|
||||||
|
"operating_system": "Ubuntu 22.04.4",
|
||||||
|
"sw_notes": ""
|
||||||
|
}
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
import os, sys, pickle, time
|
import os, sys, pickle, time, re
|
||||||
import numpy as np
|
import numpy as np
|
||||||
if "FLOAT16" not in os.environ: os.environ["FLOAT16"] = "1"
|
if "FLOAT16" not in os.environ: os.environ["FLOAT16"] = "1"
|
||||||
if "IMAGE" not in os.environ: os.environ["IMAGE"] = "2"
|
if "IMAGE" not in os.environ: os.environ["IMAGE"] = "2"
|
||||||
@@ -10,7 +10,7 @@ from tinygrad.helpers import DEBUG, getenv
|
|||||||
from tinygrad.engine.realize import CompiledRunner
|
from tinygrad.engine.realize import CompiledRunner
|
||||||
|
|
||||||
import onnx
|
import onnx
|
||||||
from tinygrad.frontend.onnx import OnnxRunner
|
from tinygrad.nn.onnx import OnnxRunner
|
||||||
|
|
||||||
OPENPILOT_MODEL = sys.argv[1] if len(sys.argv) > 1 else "https://github.com/commaai/openpilot/raw/v0.9.7/selfdrive/modeld/models/supercombo.onnx"
|
OPENPILOT_MODEL = sys.argv[1] if len(sys.argv) > 1 else "https://github.com/commaai/openpilot/raw/v0.9.7/selfdrive/modeld/models/supercombo.onnx"
|
||||||
OUTPUT = sys.argv[2] if len(sys.argv) > 2 else "/tmp/openpilot.pkl"
|
OUTPUT = sys.argv[2] if len(sys.argv) > 2 else "/tmp/openpilot.pkl"
|
||||||
@@ -52,6 +52,8 @@ def compile(onnx_file):
|
|||||||
kernel_count += 1
|
kernel_count += 1
|
||||||
read_image_count += ei.prg.p.src.count("read_image")
|
read_image_count += ei.prg.p.src.count("read_image")
|
||||||
gated_read_image_count += ei.prg.p.src.count("?read_image")
|
gated_read_image_count += ei.prg.p.src.count("?read_image")
|
||||||
|
for v in [m.group(1) for m in re.finditer(r'(val\d+)\s*=\s*read_imagef\(', ei.prg.p.src)]:
|
||||||
|
if len(re.findall(fr'[\?\:]{v}\.[xyzw]', ei.prg.p.src)) > 0: gated_read_image_count += 1
|
||||||
print(f"{kernel_count=}, {read_image_count=}, {gated_read_image_count=}")
|
print(f"{kernel_count=}, {read_image_count=}, {gated_read_image_count=}")
|
||||||
if (allowed_kernel_count:=getenv("ALLOWED_KERNEL_COUNT", -1)) != -1:
|
if (allowed_kernel_count:=getenv("ALLOWED_KERNEL_COUNT", -1)) != -1:
|
||||||
assert kernel_count == allowed_kernel_count, f"different kernels! {kernel_count=}, {allowed_kernel_count=}"
|
assert kernel_count == allowed_kernel_count, f"different kernels! {kernel_count=}, {allowed_kernel_count=}"
|
||||||
@@ -77,13 +79,20 @@ def test_vs_compile(run, new_inputs, test_val=None):
|
|||||||
**{k:Tensor(v, device="NPY").realize() for k,v in new_inputs_numpy.items() if 'img' not in k}}
|
**{k:Tensor(v, device="NPY").realize() for k,v in new_inputs_numpy.items() if 'img' not in k}}
|
||||||
|
|
||||||
# run 20 times
|
# run 20 times
|
||||||
|
step_times = []
|
||||||
for _ in range(20):
|
for _ in range(20):
|
||||||
st = time.perf_counter()
|
st = time.perf_counter()
|
||||||
out = run(**inputs)
|
out = run(**inputs)
|
||||||
mt = time.perf_counter()
|
mt = time.perf_counter()
|
||||||
val = out.numpy()
|
val = out.numpy()
|
||||||
et = time.perf_counter()
|
et = time.perf_counter()
|
||||||
print(f"enqueue {(mt-st)*1e3:6.2f} ms -- total run {(et-st)*1e3:6.2f} ms")
|
step_times.append((et-st)*1e3)
|
||||||
|
print(f"enqueue {(mt-st)*1e3:6.2f} ms -- total run {step_times[-1]:6.2f} ms")
|
||||||
|
|
||||||
|
if (assert_time:=getenv("ASSERT_MIN_STEP_TIME")):
|
||||||
|
min_time = min(step_times)
|
||||||
|
assert min_time < assert_time, f"Speed regression, expected min step time of < {assert_time} ms but took: {min_time} ms"
|
||||||
|
|
||||||
print(out, val.shape, val.dtype)
|
print(out, val.shape, val.dtype)
|
||||||
if test_val is not None: np.testing.assert_equal(test_val, val)
|
if test_val is not None: np.testing.assert_equal(test_val, val)
|
||||||
print("**** test done ****")
|
print("**** test done ****")
|
||||||
|
|||||||
@@ -1,12 +1,12 @@
|
|||||||
import sys
|
import sys
|
||||||
from tinygrad import Tensor, fetch, GlobalCounters, dtypes
|
from tinygrad import Tensor, fetch, GlobalCounters, dtypes
|
||||||
from tinygrad.uop.ops import UOp
|
from tinygrad.uop.ops import UOp
|
||||||
from tinygrad.frontend.onnx import OnnxRunner
|
from tinygrad.nn.onnx import OnnxRunner
|
||||||
from tinygrad.schedule.kernelize import get_kernelize_map
|
from tinygrad.schedule.rangeify import get_rangeify_map
|
||||||
from tinygrad.engine.schedule import create_schedule_with_vars
|
from tinygrad.engine.schedule import create_schedule_with_vars
|
||||||
from tinygrad.engine.realize import run_schedule
|
from tinygrad.engine.realize import run_schedule
|
||||||
|
|
||||||
# NOLOCALS=1 GPU=1 IMAGE=2 FLOAT16=1 VIZ=1 DEBUG=2 python3 examples/openpilot/compile4.py
|
# NOLOCALS=1 CL=1 IMAGE=2 FLOAT16=1 VIZ=1 DEBUG=2 python3 examples/openpilot/compile4.py
|
||||||
|
|
||||||
OPENPILOT_MODEL = sys.argv[1] if len(sys.argv) > 1 else "https://github.com/commaai/openpilot/raw/v0.9.7/selfdrive/modeld/models/supercombo.onnx"
|
OPENPILOT_MODEL = sys.argv[1] if len(sys.argv) > 1 else "https://github.com/commaai/openpilot/raw/v0.9.7/selfdrive/modeld/models/supercombo.onnx"
|
||||||
OUTPUT = sys.argv[2] if len(sys.argv) > 2 else "/tmp/openpilot.pkl"
|
OUTPUT = sys.argv[2] if len(sys.argv) > 2 else "/tmp/openpilot.pkl"
|
||||||
@@ -33,7 +33,7 @@ if __name__ == "__main__":
|
|||||||
if not in_target_path[s]:
|
if not in_target_path[s]:
|
||||||
independent_set[s] = None
|
independent_set[s] = None
|
||||||
independent = UOp.sink(*independent_set.keys())
|
independent = UOp.sink(*independent_set.keys())
|
||||||
kernelized = get_kernelize_map(independent)
|
kernelized = get_rangeify_map(independent)
|
||||||
independent = independent.substitute(kernelized)
|
independent = independent.substitute(kernelized)
|
||||||
schedule, var_vals = create_schedule_with_vars(independent)
|
schedule, var_vals = create_schedule_with_vars(independent)
|
||||||
run_schedule(schedule)
|
run_schedule(schedule)
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ class Model(nn.Module):
|
|||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
if getenv("TINY_BACKEND"):
|
if getenv("TINY_BACKEND"):
|
||||||
import tinygrad.frontend.torch # noqa: F401
|
import tinygrad.nn.torch # noqa: F401
|
||||||
device = torch.device("tiny")
|
device = torch.device("tiny")
|
||||||
else:
|
else:
|
||||||
device = torch.device({"METAL":"mps","NV":"cuda"}.get(Device.DEFAULT, "cpu"))
|
device = torch.device({"METAL":"mps","NV":"cuda"}.get(Device.DEFAULT, "cpu"))
|
||||||
|
|||||||
+2
-2
@@ -8,7 +8,7 @@ from typing import Dict, Union
|
|||||||
|
|
||||||
from extra.models.llama import Transformer, convert_from_huggingface, fix_bf16
|
from extra.models.llama import Transformer, convert_from_huggingface, fix_bf16
|
||||||
from examples.llama3 import load
|
from examples.llama3 import load
|
||||||
from tinygrad import nn, Tensor
|
from tinygrad import nn, Tensor, Device
|
||||||
from tinygrad.helpers import fetch, colored, GlobalCounters, Timing, DEBUG
|
from tinygrad.helpers import fetch, colored, GlobalCounters, Timing, DEBUG
|
||||||
from tinygrad.nn.state import load_state_dict, get_parameters
|
from tinygrad.nn.state import load_state_dict, get_parameters
|
||||||
|
|
||||||
@@ -80,7 +80,7 @@ if __name__ == "__main__":
|
|||||||
st = GlobalCounters.time_sum_s
|
st = GlobalCounters.time_sum_s
|
||||||
next_tok = Tensor([toks[start_pos:]]) if tok_tensor is None or (len(toks)-start_pos) > 1 else tok_tensor.reshape(1, 1)
|
next_tok = Tensor([toks[start_pos:]]) if tok_tensor is None or (len(toks)-start_pos) > 1 else tok_tensor.reshape(1, 1)
|
||||||
with Timing("total ", enabled=args.timing, on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
with Timing("total ", enabled=args.timing, on_exit=lambda x: f", {1e9/x:.2f} tok/s, {GlobalCounters.global_mem/x:.2f} GB/s, param {param_bytes/x:.2f} GB/s"):
|
||||||
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on GPU" if DEBUG>=2 else "") +
|
with Timing("enqueue in ", on_exit=(lambda et: (f", {(GlobalCounters.time_sum_s-st)*1e3:.2f} ms on {Device.DEFAULT}" if DEBUG>=2 else "") +
|
||||||
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB" +
|
f", {GlobalCounters.global_ops*1e-9:.2f} GOPS, {GlobalCounters.global_mem*1e-9:.2f} GB" +
|
||||||
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=args.timing):
|
(f", {GlobalCounters.global_mem*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s, param {param_bytes*1e-9/(GlobalCounters.time_sum_s-st):.2f} GB/s" if DEBUG>=2 else "")) if DEBUG else None, enabled=args.timing):
|
||||||
tok_tensor = transformer(next_tok, start_pos, args.temperature)
|
tok_tensor = transformer(next_tok, start_pos, args.temperature)
|
||||||
|
|||||||
+11
-4
@@ -6,7 +6,7 @@
|
|||||||
from tinygrad import Tensor, TinyJit, dtypes, GlobalCounters
|
from tinygrad import Tensor, TinyJit, dtypes, GlobalCounters
|
||||||
from tinygrad.nn import Conv2d, GroupNorm
|
from tinygrad.nn import Conv2d, GroupNorm
|
||||||
from tinygrad.nn.state import safe_load, load_state_dict
|
from tinygrad.nn.state import safe_load, load_state_dict
|
||||||
from tinygrad.helpers import fetch, trange, colored, Timing
|
from tinygrad.helpers import fetch, trange, colored, Timing, getenv
|
||||||
from extra.models.clip import Embedder, FrozenClosedClipEmbedder, FrozenOpenClipEmbedder
|
from extra.models.clip import Embedder, FrozenClosedClipEmbedder, FrozenOpenClipEmbedder
|
||||||
from extra.models.unet import UNetModel, Upsample, Downsample, timestep_embedding
|
from extra.models.unet import UNetModel, Upsample, Downsample, timestep_embedding
|
||||||
from extra.bench_log import BenchEvent, WallTimeEvent
|
from extra.bench_log import BenchEvent, WallTimeEvent
|
||||||
@@ -14,7 +14,7 @@ from examples.stable_diffusion import ResnetBlock, Mid
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
||||||
from typing import Dict, List, Callable, Optional, Any, Set, Tuple, Union, Type
|
from typing import Dict, List, Callable, Optional, Any, Set, Tuple, Union, Type
|
||||||
import argparse, tempfile
|
import argparse, tempfile, time
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
@@ -342,11 +342,13 @@ class DPMPP2MSampler:
|
|||||||
sigmas = self.discretization(num_steps).to(x.device)
|
sigmas = self.discretization(num_steps).to(x.device)
|
||||||
x *= Tensor.sqrt(1.0 + sigmas[0] ** 2.0)
|
x *= Tensor.sqrt(1.0 + sigmas[0] ** 2.0)
|
||||||
num_sigmas = len(sigmas)
|
num_sigmas = len(sigmas)
|
||||||
|
step_times = []
|
||||||
|
|
||||||
old_denoised = None
|
old_denoised = None
|
||||||
for i in trange(num_sigmas - 1):
|
for i in trange(num_sigmas - 1):
|
||||||
with Timing("step in ", enabled=timing, on_exit=lambda _: f", using {GlobalCounters.mem_used/1e9:.2f} GB"):
|
with Timing("step in ", enabled=timing, on_exit=lambda _: f", using {GlobalCounters.mem_used/1e9:.2f} GB"):
|
||||||
GlobalCounters.reset()
|
GlobalCounters.reset()
|
||||||
|
st = time.perf_counter_ns()
|
||||||
with WallTimeEvent(BenchEvent.STEP):
|
with WallTimeEvent(BenchEvent.STEP):
|
||||||
x, old_denoised = self.sampler_step(
|
x, old_denoised = self.sampler_step(
|
||||||
old_denoised=old_denoised,
|
old_denoised=old_denoised,
|
||||||
@@ -358,8 +360,13 @@ class DPMPP2MSampler:
|
|||||||
c=c,
|
c=c,
|
||||||
uc=uc,
|
uc=uc,
|
||||||
)
|
)
|
||||||
|
step_times.append(t:=(time.perf_counter_ns() - st)*1e-6)
|
||||||
x.realize(old_denoised)
|
x.realize(old_denoised)
|
||||||
|
|
||||||
|
if (assert_time:=getenv("ASSERT_MIN_STEP_TIME")):
|
||||||
|
min_time = min(step_times)
|
||||||
|
assert min_time < assert_time, f"Speed regression, expected min step time of < {assert_time} ms but took: {min_time} ms"
|
||||||
|
|
||||||
return x
|
return x
|
||||||
|
|
||||||
|
|
||||||
@@ -430,8 +437,8 @@ if __name__ == "__main__":
|
|||||||
im.show()
|
im.show()
|
||||||
|
|
||||||
# validation!
|
# validation!
|
||||||
if args.prompt == default_prompt and args.steps == 10 and args.seed == 0 and args.guidance == 6.0 and args.width == args.height == 1024 \
|
is_default = args.prompt == default_prompt and args.steps == 10 and args.seed == 0 and args.guidance == 6.0 and args.width == args.height == 1024
|
||||||
and not args.weights:
|
if is_default and not args.weights and not args.fakeweights:
|
||||||
ref_image = Tensor(np.array(Image.open(Path(__file__).parent / "sdxl_seed0.png")))
|
ref_image = Tensor(np.array(Image.open(Path(__file__).parent / "sdxl_seed0.png")))
|
||||||
distance = (((x.cast(dtypes.float) - ref_image.cast(dtypes.float)) / ref_image.max())**2).mean().item()
|
distance = (((x.cast(dtypes.float) - ref_image.cast(dtypes.float)) / ref_image.max())**2).mean().item()
|
||||||
assert distance < 4e-3, colored(f"validation failed with {distance=}", "red")
|
assert distance < 4e-3, colored(f"validation failed with {distance=}", "red")
|
||||||
|
|||||||
@@ -2,18 +2,20 @@
|
|||||||
# https://github.com/ekagra-ranjan/huggingface-blog/blob/main/stable_diffusion.md
|
# https://github.com/ekagra-ranjan/huggingface-blog/blob/main/stable_diffusion.md
|
||||||
import tempfile
|
import tempfile
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import argparse
|
import argparse, time
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
from typing import Dict, Any
|
from typing import Dict, Any
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from tinygrad import Device, GlobalCounters, dtypes, Tensor, TinyJit
|
from tinygrad import Device, GlobalCounters, dtypes, Tensor, TinyJit
|
||||||
from tinygrad.helpers import Timing, Context, getenv, fetch, colored, tqdm
|
from tinygrad.helpers import Timing, Context, getenv, fetch, colored, tqdm, flatten
|
||||||
from tinygrad.nn import Conv2d, GroupNorm
|
from tinygrad.nn import Conv2d, GroupNorm
|
||||||
from tinygrad.nn.state import torch_load, load_state_dict, get_state_dict
|
from tinygrad.nn.state import torch_load, load_state_dict, get_state_dict
|
||||||
from extra.models.clip import Closed, Tokenizer
|
from extra.models.clip import Closed, Tokenizer, FrozenOpenClipEmbedder
|
||||||
|
from extra.models import unet, clip
|
||||||
from extra.models.unet import UNetModel
|
from extra.models.unet import UNetModel
|
||||||
|
from examples.mlperf.initializers import AutocastLinear, AutocastConv2d, AutocastGroupNorm, AutocastLayerNorm, zero_module, attn_f32_softmax, gelu_erf
|
||||||
from extra.bench_log import BenchEvent, WallTimeEvent
|
from extra.bench_log import BenchEvent, WallTimeEvent
|
||||||
|
|
||||||
class AttnBlock:
|
class AttnBlock:
|
||||||
@@ -154,12 +156,46 @@ unet_params: Dict[str,Any] = {
|
|||||||
"use_linear": False,
|
"use_linear": False,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
mlperf_params: Dict[str,Any] = {"adm_in_ch": None, "in_ch": 4, "out_ch": 4, "model_ch": 320, "attention_resolutions": [4, 2, 1], "num_res_blocks": 2,
|
||||||
|
"channel_mult": [1, 2, 4, 4], "d_head": 64, "transformer_depth": [1, 1, 1, 1], "ctx_dim": 1024, "use_linear": True,
|
||||||
|
"num_groups":16, "st_norm_eps":1e-6}
|
||||||
|
|
||||||
class StableDiffusion:
|
class StableDiffusion:
|
||||||
def __init__(self):
|
def __init__(self, version:str|None=None, pretrained:str|None=None):
|
||||||
self.alphas_cumprod = get_alphas_cumprod()
|
self.alphas_cumprod = get_alphas_cumprod()
|
||||||
self.model = namedtuple("DiffusionModel", ["diffusion_model"])(diffusion_model = UNetModel(**unet_params))
|
if version != "v2-mlperf-train":
|
||||||
self.first_stage_model = AutoencoderKL()
|
self.first_stage_model = AutoencoderKL() # only needed for decoding generated latents to images; not needed in mlperf training from preprocessed moments
|
||||||
self.cond_stage_model = namedtuple("CondStageModel", ["transformer"])(transformer = namedtuple("Transformer", ["text_model"])(text_model = Closed.ClipTextTransformer()))
|
|
||||||
|
if not version:
|
||||||
|
self.cond_stage_model = namedtuple("CondStageModel", ["transformer"])(transformer = namedtuple("Transformer", ["text_model"])(text_model = Closed.ClipTextTransformer()))
|
||||||
|
unet_init_params = unet_params
|
||||||
|
elif version in {"v2-mlperf-train", "v2-mlperf-eval"}:
|
||||||
|
unet_init_params = mlperf_params
|
||||||
|
clip.gelu = gelu_erf
|
||||||
|
self.cond_stage_model = FrozenOpenClipEmbedder(**{"dims": 1024, "n_heads": 16, "layers": 24, "return_pooled": False, "ln_penultimate": True,
|
||||||
|
"clip_tokenizer_version": "sd_mlperf_v5_0"})
|
||||||
|
unet.Linear, unet.Conv2d, unet.GroupNorm, unet.LayerNorm = AutocastLinear, AutocastConv2d, AutocastGroupNorm, AutocastLayerNorm
|
||||||
|
unet.attention, unet.gelu, unet.mixed_precision_dtype = attn_f32_softmax, gelu_erf, dtypes.bfloat16
|
||||||
|
if pretrained:
|
||||||
|
print("loading text encoder")
|
||||||
|
weights: dict[str,Tensor] = {k.replace("cond_stage_model.", "", 1):v for k,v in torch_load(pretrained)["state_dict"].items() if k.startswith("cond_stage_model.")}
|
||||||
|
weights["model.attn_mask"] = Tensor.full((77, 77), fill_value=float("-inf")).triu(1)
|
||||||
|
load_state_dict(self.cond_stage_model, weights)
|
||||||
|
# only the eval model needs the decoder
|
||||||
|
if version == "v2-mlperf-eval":
|
||||||
|
print("loading image latent encoder")
|
||||||
|
weights = {k.replace("first_stage_model.", "", 1):v for k,v in torch_load(pretrained)["state_dict"].items() if k.startswith("first_stage_model.")}
|
||||||
|
load_state_dict(self.first_stage_model, weights)
|
||||||
|
|
||||||
|
self.model = namedtuple("DiffusionModel", ["diffusion_model"])(diffusion_model = UNetModel(**unet_init_params))
|
||||||
|
if version == "v2-mlperf-train":
|
||||||
|
# the mlperf reference inits certain weights as zeroes
|
||||||
|
for bb in flatten(self.model.diffusion_model.input_blocks) + self.model.diffusion_model.middle_block + flatten(self.model.diffusion_model.output_blocks):
|
||||||
|
if isinstance(bb, unet.ResBlock):
|
||||||
|
zero_module(bb.out_layers[3])
|
||||||
|
elif isinstance(bb, unet.SpatialTransformer):
|
||||||
|
zero_module(bb.proj_out)
|
||||||
|
zero_module(self.model.diffusion_model.out[2])
|
||||||
|
|
||||||
def get_x_prev_and_pred_x0(self, x, e_t, a_t, a_prev):
|
def get_x_prev_and_pred_x0(self, x, e_t, a_t, a_prev):
|
||||||
temperature = 1
|
temperature = 1
|
||||||
@@ -233,12 +269,14 @@ if __name__ == "__main__":
|
|||||||
|
|
||||||
# load in weights
|
# load in weights
|
||||||
with WallTimeEvent(BenchEvent.LOAD_WEIGHTS):
|
with WallTimeEvent(BenchEvent.LOAD_WEIGHTS):
|
||||||
load_state_dict(model, torch_load(fetch('https://huggingface.co/CompVis/stable-diffusion-v-1-4-original/resolve/main/sd-v1-4.ckpt', 'sd-v1-4.ckpt'))['state_dict'], strict=False)
|
load_state_dict(model, torch_load(fetch('https://huggingface.co/CompVis/stable-diffusion-v-1-4-original/resolve/main/sd-v1-4.ckpt', 'sd-v1-4.ckpt'))['state_dict'], verbose=False, strict=False, realize=False)
|
||||||
|
|
||||||
if args.fp16:
|
if args.fp16:
|
||||||
for k,v in get_state_dict(model).items():
|
for k,v in get_state_dict(model).items():
|
||||||
if k.startswith("model"):
|
if k.startswith("model"):
|
||||||
v.replace(v.cast(dtypes.float16).realize())
|
v.replace(v.cast(dtypes.float16))
|
||||||
|
|
||||||
|
Tensor.realize(*get_state_dict(model).values())
|
||||||
|
|
||||||
# run through CLIP to get context
|
# run through CLIP to get context
|
||||||
tokenizer = Tokenizer.ClipTokenizer()
|
tokenizer = Tokenizer.ClipTokenizer()
|
||||||
@@ -266,17 +304,23 @@ if __name__ == "__main__":
|
|||||||
def run(model, *x): return model(*x).realize()
|
def run(model, *x): return model(*x).realize()
|
||||||
|
|
||||||
# this is diffusion
|
# this is diffusion
|
||||||
|
step_times = []
|
||||||
with Context(BEAM=getenv("LATEBEAM")):
|
with Context(BEAM=getenv("LATEBEAM")):
|
||||||
for index, timestep in (t:=tqdm(list(enumerate(timesteps))[::-1])):
|
for index, timestep in (t:=tqdm(list(enumerate(timesteps))[::-1])):
|
||||||
GlobalCounters.reset()
|
GlobalCounters.reset()
|
||||||
|
st = time.perf_counter_ns()
|
||||||
t.set_description("%3d %3d" % (index, timestep))
|
t.set_description("%3d %3d" % (index, timestep))
|
||||||
with Timing("step in ", enabled=args.timing, on_exit=lambda _: f", using {GlobalCounters.mem_used/1e9:.2f} GB"):
|
with Timing("step in ", enabled=args.timing, on_exit=lambda _: f", using {GlobalCounters.mem_used/1e9:.2f} GB"):
|
||||||
with WallTimeEvent(BenchEvent.STEP):
|
with WallTimeEvent(BenchEvent.STEP):
|
||||||
tid = Tensor([index])
|
tid = Tensor([index])
|
||||||
latent = run(model, unconditional_context, context, latent, Tensor([timestep]), alphas[tid], alphas_prev[tid], Tensor([args.guidance]))
|
latent = run(model, unconditional_context, context, latent, Tensor([timestep]), alphas[tid], alphas_prev[tid], Tensor([args.guidance]))
|
||||||
if args.timing: Device[Device.DEFAULT].synchronize()
|
if args.timing: Device[Device.DEFAULT].synchronize()
|
||||||
|
step_times.append((time.perf_counter_ns() - st)*1e-6)
|
||||||
del run
|
del run
|
||||||
|
|
||||||
|
if (assert_time:=getenv("ASSERT_MIN_STEP_TIME")):
|
||||||
|
min_time = min(step_times)
|
||||||
|
assert min_time < assert_time, f"Speed regression, expected min step time of < {assert_time} ms but took: {min_time} ms"
|
||||||
# upsample latent space to image with autoencoder
|
# upsample latent space to image with autoencoder
|
||||||
x = model.decode(latent)
|
x = model.decode(latent)
|
||||||
print(x.shape)
|
print(x.shape)
|
||||||
|
|||||||
@@ -32,7 +32,7 @@ if __name__ == "__main__":
|
|||||||
|
|
||||||
lr = 5e-3
|
lr = 5e-3
|
||||||
transform = ComposeTransforms([
|
transform = ComposeTransforms([
|
||||||
lambda x: [Image.fromarray(xx, mode='L').resize((64, 64)) for xx in x],
|
lambda x: [Image.fromarray(xx).resize((64, 64)) for xx in x],
|
||||||
lambda x: np.stack([np.asarray(xx) for xx in x], 0),
|
lambda x: np.stack([np.asarray(xx) for xx in x], 0),
|
||||||
lambda x: x / 255.0,
|
lambda x: x / 255.0,
|
||||||
lambda x: np.tile(np.expand_dims(x, 1), (1, 3, 1, 1)).astype(np.float32),
|
lambda x: np.tile(np.expand_dims(x, 1), (1, 3, 1, 1)).astype(np.float32),
|
||||||
|
|||||||
+1
-1
@@ -109,7 +109,7 @@ class TextDecoder:
|
|||||||
|
|
||||||
def forward(self, x:Tensor, pos:Union[Variable, Literal[0]], encoded_audio:Tensor):
|
def forward(self, x:Tensor, pos:Union[Variable, Literal[0]], encoded_audio:Tensor):
|
||||||
seqlen = x.shape[-1]
|
seqlen = x.shape[-1]
|
||||||
x = self.token_embedding(x) + self.positional_embedding.shrink(((pos, pos+seqlen), None, None))
|
x = self.token_embedding(x) + self.positional_embedding.shrink(((pos, pos+seqlen), None))
|
||||||
for block in self.blocks: x = block(x, xa=encoded_audio, mask=self.mask, len=pos)
|
for block in self.blocks: x = block(x, xa=encoded_audio, mask=self.mask, len=pos)
|
||||||
return self.output_tok(x)
|
return self.output_tok(x)
|
||||||
|
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
import os
|
import os
|
||||||
from ultralytics import YOLO
|
from ultralytics import YOLO
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from tinygrad.frontend.onnx import OnnxRunner
|
from tinygrad.nn.onnx import OnnxRunner
|
||||||
from extra.onnx_helpers import get_example_inputs
|
from extra.onnx_helpers import get_example_inputs
|
||||||
|
|
||||||
os.chdir("/tmp")
|
os.chdir("/tmp")
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ def main():
|
|||||||
dev = PCIIface(None, 0)
|
dev = PCIIface(None, 0)
|
||||||
for x, y in dev.dev_impl.__dict__.items():
|
for x, y in dev.dev_impl.__dict__.items():
|
||||||
if isinstance(y, AMRegister):
|
if isinstance(y, AMRegister):
|
||||||
for inst, addr in y.addr.keys(): reg_names[addr] = f"{x}, xcc={inst}"
|
for inst, addr in y.addr.items(): reg_names[addr] = f"{x}, xcc={inst}"
|
||||||
|
|
||||||
with open(sys.argv[1], 'r') as f:
|
with open(sys.argv[1], 'r') as f:
|
||||||
log_content = log_content_them = f.read()
|
log_content = log_content_them = f.read()
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
# copying the kernels from https://github.com/microsoft/ArchProbe into Python
|
# copying the kernels from https://github.com/microsoft/ArchProbe into Python
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pickle
|
import pickle
|
||||||
from tinygrad.runtime.ops_gpu import CLProgram, CLBuffer
|
from tinygrad.runtime.ops_cl import CLProgram, CLBuffer
|
||||||
from tinygrad import dtypes
|
from tinygrad import dtypes
|
||||||
from tqdm import trange, tqdm
|
from tqdm import trange, tqdm
|
||||||
from matplotlib import pyplot as plt
|
from matplotlib import pyplot as plt
|
||||||
|
|||||||
@@ -4,7 +4,7 @@ from tinygrad import dtypes
|
|||||||
from tinygrad.codegen.assembly import AssemblyCodegen, Register
|
from tinygrad.codegen.assembly import AssemblyCodegen, Register
|
||||||
from tinygrad.codegen.opt.kernel import Ops
|
from tinygrad.codegen.opt.kernel import Ops
|
||||||
from tinygrad.uop.ops import BinaryOps, UnaryOps, TernaryOps
|
from tinygrad.uop.ops import BinaryOps, UnaryOps, TernaryOps
|
||||||
from tinygrad.runtime.ops_gpu import ROCM_LLVM_PATH
|
from tinygrad.runtime.ops_cl import ROCM_LLVM_PATH
|
||||||
|
|
||||||
# ugh, is this really needed?
|
# ugh, is this really needed?
|
||||||
from extra.helpers import enable_early_exec
|
from extra.helpers import enable_early_exec
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ from tinygrad.helpers import colored
|
|||||||
from extra.helpers import enable_early_exec
|
from extra.helpers import enable_early_exec
|
||||||
early_exec = enable_early_exec()
|
early_exec = enable_early_exec()
|
||||||
|
|
||||||
from tinygrad.runtime.ops_gpu import CLProgram, CLBuffer, ROCM_LLVM_PATH
|
from tinygrad.runtime.ops_cl import CLProgram, CLBuffer, ROCM_LLVM_PATH
|
||||||
|
|
||||||
ENABLE_NON_ASM = False
|
ENABLE_NON_ASM = False
|
||||||
|
|
||||||
|
|||||||
@@ -10,13 +10,13 @@ from tinygrad.renderer.cstyle import ClangRenderer
|
|||||||
render_dtype = ClangRenderer().render_dtype
|
render_dtype = ClangRenderer().render_dtype
|
||||||
|
|
||||||
class ClangGraph(GraphRunner):
|
class ClangGraph(GraphRunner):
|
||||||
def __init__(self, jit_cache: List[ExecItem], input_rawbuffers: List[Buffer], var_vals: Dict[Variable, int]):
|
def __init__(self, jit_cache: List[ExecItem], input_rawbuffers: List[Buffer], var_vals: Dict[str, int]):
|
||||||
super().__init__(jit_cache, input_rawbuffers, var_vals)
|
super().__init__(jit_cache, input_rawbuffers, var_vals)
|
||||||
if not all(isinstance(ji.prg, CompiledRunner) for ji in jit_cache): raise GraphException
|
if not all(isinstance(ji.prg, CompiledRunner) for ji in jit_cache): raise GraphException
|
||||||
|
|
||||||
prgs = '\n'.join(dedup([cast(CompiledRunner, ji.prg).p.src for ji in jit_cache]))
|
prgs = '\n'.join(dedup([cast(CompiledRunner, ji.prg).p.src for ji in jit_cache]))
|
||||||
args = [f"{render_dtype(x.dtype)}* arg{i}" for i,x in enumerate(input_rawbuffers)]
|
args = [f"{render_dtype(x.dtype)}* arg{i}" for i,x in enumerate(input_rawbuffers)]
|
||||||
args += sorted([f"int {v.expr}" for v in var_vals])
|
args += sorted([f"int {v}" for v in var_vals])
|
||||||
code = ["void batched("+','.join(args)+") {"]
|
code = ["void batched("+','.join(args)+") {"]
|
||||||
for ji in jit_cache:
|
for ji in jit_cache:
|
||||||
args = []
|
args = []
|
||||||
@@ -34,6 +34,6 @@ class ClangGraph(GraphRunner):
|
|||||||
assert compiler is not None
|
assert compiler is not None
|
||||||
self._prg = ClangProgram("batched", compiler.compile(prgs+"\n"+"\n".join(code))) # no point in caching the pointers
|
self._prg = ClangProgram("batched", compiler.compile(prgs+"\n"+"\n".join(code))) # no point in caching the pointers
|
||||||
|
|
||||||
def __call__(self, rawbufs: List[Buffer], var_vals: Dict[Variable, int], wait=False):
|
def __call__(self, rawbufs: List[Buffer], var_vals: Dict[str, int], wait=False):
|
||||||
return cpu_time_execution(
|
return cpu_time_execution(
|
||||||
lambda: self._prg(*[x._buf for x in rawbufs], *[x[1] for x in sorted(var_vals.items(), key=lambda x: x[0].expr)]), enable=wait)
|
lambda: self._prg(*[x._buf for x in rawbufs], *[x[1] for x in sorted(var_vals.items(), key=lambda x: x[0])]), enable=wait)
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ class VirtAQLQueue(AQLQueue):
|
|||||||
self.available_packet_slots -= 1
|
self.available_packet_slots -= 1
|
||||||
|
|
||||||
class HSAGraph(MultiGraphRunner):
|
class HSAGraph(MultiGraphRunner):
|
||||||
def __init__(self, jit_cache: List[ExecItem], input_rawbuffers: List[Buffer], var_vals: Dict[Variable, int]):
|
def __init__(self, jit_cache: List[ExecItem], input_rawbuffers: List[Buffer], var_vals: Dict[str, int]):
|
||||||
super().__init__(jit_cache, input_rawbuffers, var_vals)
|
super().__init__(jit_cache, input_rawbuffers, var_vals)
|
||||||
|
|
||||||
# Check all jit items are compatible.
|
# Check all jit items are compatible.
|
||||||
@@ -53,7 +53,7 @@ class HSAGraph(MultiGraphRunner):
|
|||||||
self.ji_kargs_structs[j] = ji.prg._prg.args_struct_t.from_address(kernargs_ptrs[ji.prg.dev])
|
self.ji_kargs_structs[j] = ji.prg._prg.args_struct_t.from_address(kernargs_ptrs[ji.prg.dev])
|
||||||
kernargs_ptrs[ji.prg.dev] += round_up(ctypes.sizeof(ji.prg._prg.args_struct_t), 16)
|
kernargs_ptrs[ji.prg.dev] += round_up(ctypes.sizeof(ji.prg._prg.args_struct_t), 16)
|
||||||
for i in range(len(ji.bufs)): self.ji_kargs_structs[j].__setattr__(f'f{i}', cast(Buffer, ji.bufs[i])._buf)
|
for i in range(len(ji.bufs)): self.ji_kargs_structs[j].__setattr__(f'f{i}', cast(Buffer, ji.bufs[i])._buf)
|
||||||
for i in range(len(ji.prg.p.vars)): self.ji_kargs_structs[j].__setattr__(f'v{i}', var_vals[ji.prg.p.vars[i]])
|
for i in range(len(ji.prg.p.vars)): self.ji_kargs_structs[j].__setattr__(f'v{i}', var_vals[ji.prg.p.vars[i].expr])
|
||||||
|
|
||||||
# Build queues.
|
# Build queues.
|
||||||
self.virt_aql_queues: Dict[Compiled, VirtAQLQueue] = {dev:VirtAQLQueue(dev, 2*len(self.jit_cache)+16) for dev in self.devices}
|
self.virt_aql_queues: Dict[Compiled, VirtAQLQueue] = {dev:VirtAQLQueue(dev, 2*len(self.jit_cache)+16) for dev in self.devices}
|
||||||
@@ -106,7 +106,7 @@ class HSAGraph(MultiGraphRunner):
|
|||||||
for sig in self.signals_to_reset: hsa.hsa_signal_silent_store_relaxed(sig, 0)
|
for sig in self.signals_to_reset: hsa.hsa_signal_silent_store_relaxed(sig, 0)
|
||||||
hsa.hsa_signal_silent_store_relaxed(self.finish_signal, 0)
|
hsa.hsa_signal_silent_store_relaxed(self.finish_signal, 0)
|
||||||
|
|
||||||
def __call__(self, input_rawbuffers: List[Buffer], var_vals: Dict[Variable, int], wait=False) -> Optional[float]:
|
def __call__(self, input_rawbuffers: List[Buffer], var_vals: Dict[str, int], wait=False) -> Optional[float]:
|
||||||
# Wait and restore signals
|
# Wait and restore signals
|
||||||
hsa.hsa_signal_wait_scacquire(self.finish_signal, hsa.HSA_SIGNAL_CONDITION_LT, 1, (1 << 64) - 1, hsa.HSA_WAIT_STATE_ACTIVE)
|
hsa.hsa_signal_wait_scacquire(self.finish_signal, hsa.HSA_SIGNAL_CONDITION_LT, 1, (1 << 64) - 1, hsa.HSA_WAIT_STATE_ACTIVE)
|
||||||
for sig in self.signals_to_reset: hsa.hsa_signal_silent_store_relaxed(sig, 1)
|
for sig in self.signals_to_reset: hsa.hsa_signal_silent_store_relaxed(sig, 1)
|
||||||
@@ -123,7 +123,7 @@ class HSAGraph(MultiGraphRunner):
|
|||||||
# Update var_vals
|
# Update var_vals
|
||||||
for j in self.jc_idx_with_updatable_var_vals:
|
for j in self.jc_idx_with_updatable_var_vals:
|
||||||
for i,v in enumerate(cast(CompiledRunner, self.jit_cache[j].prg).p.vars):
|
for i,v in enumerate(cast(CompiledRunner, self.jit_cache[j].prg).p.vars):
|
||||||
self.ji_kargs_structs[j].__setattr__(f'v{i}', var_vals[v])
|
self.ji_kargs_structs[j].__setattr__(f'v{i}', var_vals[v.expr])
|
||||||
|
|
||||||
# Update launch dims
|
# Update launch dims
|
||||||
for j in self.jc_idx_with_updatable_launch_dims:
|
for j in self.jc_idx_with_updatable_launch_dims:
|
||||||
|
|||||||
@@ -29,10 +29,10 @@ def uops_to_rdna(function_name:str, uops:UOpGraph) -> str:
|
|||||||
r: Dict[UOp, str] = {}
|
r: Dict[UOp, str] = {}
|
||||||
for u in uops:
|
for u in uops:
|
||||||
if u.uop == UOps.SPECIAL:
|
if u.uop == UOps.SPECIAL:
|
||||||
if u.arg[1].startswith("lidx"):
|
if u.arg.startswith("lidx"):
|
||||||
r[u] = f'v{u.arg[0]}'
|
r[u] = f'v{u.src[0].arg}'
|
||||||
elif u.arg[1].startswith("gidx"):
|
elif u.arg.startswith("gidx"):
|
||||||
r[u] = f's{2+u.arg[0]}'
|
r[u] = f's{2+u.src[0].arg}'
|
||||||
else:
|
else:
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
elif u.uop == UOps.CONST:
|
elif u.uop == UOps.CONST:
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ from tinygrad.uop.ops import Ops
|
|||||||
import json
|
import json
|
||||||
from collections import OrderedDict
|
from collections import OrderedDict
|
||||||
|
|
||||||
EXPORT_SUPPORTED_DEVICE = ["WEBGPU", "CPU", "CUDA", "GPU"]
|
EXPORT_SUPPORTED_DEVICE = ["WEBGPU", "CPU", "CUDA", "CL"]
|
||||||
|
|
||||||
def compile_net(run:TinyJit, special_names:Dict[int,str]) -> Tuple[Dict[str,str],List[Tuple[str,List[str],List[int]]],Dict[str,Tuple[int,DType,int]],Dict[str,Tensor]]:
|
def compile_net(run:TinyJit, special_names:Dict[int,str]) -> Tuple[Dict[str,str],List[Tuple[str,List[str],List[int]]],Dict[str,Tuple[int,DType,int]],Dict[str,Tensor]]:
|
||||||
functions, bufs, bufs_to_save, statements, bufnum = {}, {}, {}, [], 0
|
functions, bufs, bufs_to_save, statements, bufnum = {}, {}, {}, [], 0
|
||||||
@@ -67,11 +67,12 @@ def export_model_clang(functions:Dict[str,str], statements:Dict[str,Tuple[str,in
|
|||||||
forward_args = ",".join(f"{dtype}{'*' if name not in symbolic_vars.values() else ''} {name}" for name,dtype,_ in (outputs+inputs if wasm else inputs+outputs))
|
forward_args = ",".join(f"{dtype}{'*' if name not in symbolic_vars.values() else ''} {name}" for name,dtype,_ in (outputs+inputs if wasm else inputs+outputs))
|
||||||
|
|
||||||
if not wasm:
|
if not wasm:
|
||||||
|
thread_id = 0 # NOTE: export does not support threading, thread_id is always 0
|
||||||
for name,cl in bufs_to_save.items():
|
for name,cl in bufs_to_save.items():
|
||||||
weight = ''.join(["\\x%02X"%x for x in bytes(to_mv(cl._buf.va_addr, cl._buf.size))])
|
weight = ''.join(["\\x%02X"%x for x in bytes(to_mv(cl._buf.va_addr, cl._buf.size))])
|
||||||
cprog.append(f"unsigned char {name}_data[] = \"{weight}\";")
|
cprog.append(f"unsigned char {name}_data[] = \"{weight}\";")
|
||||||
cprog += [f"{dtype_map[dtype]} {name}[{len}];" if name not in bufs_to_save else f"{dtype_map[dtype]} *{name} = ({dtype_map[dtype]} *){name}_data;" for name,(len,dtype,_key) in bufs.items() if name not in input_names+output_names]
|
cprog += [f"{dtype_map[dtype]} {name}[{len}];" if name not in bufs_to_save else f"{dtype_map[dtype]} *{name} = ({dtype_map[dtype]} *){name}_data;" for name,(len,dtype,_key) in bufs.items() if name not in input_names+output_names]
|
||||||
cprog += [f"void net({forward_args}) {{"] + [f"{name}({', '.join(args)});" for (name, args, _global_size, _local_size) in statements] + ["}"]
|
cprog += [f"void net({forward_args}) {{"] + [f"{name}({', '.join(args)}, {thread_id});" for (name, args, _global_size, _local_size) in statements] + ["}"]
|
||||||
return '\n'.join(headers + cprog)
|
return '\n'.join(headers + cprog)
|
||||||
else:
|
else:
|
||||||
if bufs_to_save:
|
if bufs_to_save:
|
||||||
@@ -239,7 +240,9 @@ export default {model_name};
|
|||||||
|
|
||||||
def export_model(model, target:str, *inputs, model_name: Optional[str] = "model", stream_weights=False):
|
def export_model(model, target:str, *inputs, model_name: Optional[str] = "model", stream_weights=False):
|
||||||
assert Device.DEFAULT in EXPORT_SUPPORTED_DEVICE, f"only {', '.join(EXPORT_SUPPORTED_DEVICE)} are supported"
|
assert Device.DEFAULT in EXPORT_SUPPORTED_DEVICE, f"only {', '.join(EXPORT_SUPPORTED_DEVICE)} are supported"
|
||||||
with Context(JIT=2): run,special_names = jit_model(model, *inputs)
|
|
||||||
|
# NOTE: CPU_COUNT=1, since export does not support threading
|
||||||
|
with Context(JIT=2, CPU_COUNT=1): run,special_names = jit_model(model, *inputs)
|
||||||
functions, statements, bufs, bufs_to_save = compile_net(run, special_names)
|
functions, statements, bufs, bufs_to_save = compile_net(run, special_names)
|
||||||
state = get_state_dict(model)
|
state = get_state_dict(model)
|
||||||
weight_names = {id(x.uop.base.realized): name for name, x in state.items()}
|
weight_names = {id(x.uop.base.realized): name for name, x in state.items()}
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ from tinygrad.dtype import AddrSpace
|
|||||||
from tinygrad.helpers import getenv, colored, prod, unwrap
|
from tinygrad.helpers import getenv, colored, prod, unwrap
|
||||||
from tinygrad.shape.shapetracker import ShapeTracker, View
|
from tinygrad.shape.shapetracker import ShapeTracker, View
|
||||||
from tinygrad.shape.view import strides_for_shape
|
from tinygrad.shape.view import strides_for_shape
|
||||||
from tinygrad.codegen.opt.kernel import axis_colors
|
from tinygrad.codegen.opt.kernel import axis_colors, Opt, OptOps
|
||||||
from tinygrad.codegen.opt.swizzler import merge_views, view_left
|
from tinygrad.codegen.opt.swizzler import merge_views, view_left
|
||||||
|
|
||||||
def to_colored(full_shape, axis_types): return '_'.join([colored(str(s), axis_colors[at]) for s,at in zip(full_shape, axis_types)])
|
def to_colored(full_shape, axis_types): return '_'.join([colored(str(s), axis_colors[at]) for s,at in zip(full_shape, axis_types)])
|
||||||
@@ -44,13 +44,27 @@ pm = PatternMatcher([
|
|||||||
(UPat(Ops.VIEW, src=(UPat(Ops.REDUCE_AXIS, src=(UPat.var("src"),), name="r"),), name="view"), swizzle_reduceop),
|
(UPat(Ops.VIEW, src=(UPat(Ops.REDUCE_AXIS, src=(UPat.var("src"),), name="r"),), name="view"), swizzle_reduceop),
|
||||||
])
|
])
|
||||||
|
|
||||||
|
def rangeify_kernel3():
|
||||||
|
a = Tensor.empty(N,N)
|
||||||
|
b = Tensor.empty(N,N)
|
||||||
|
c = a@b
|
||||||
|
#c = c.reshape((32,2,16,4,32,2,16,4)).contiguous()
|
||||||
|
sink = c.schedule()[-1].ast
|
||||||
|
#print(sink)
|
||||||
|
|
||||||
|
opts = [Opt(OptOps.UPCAST, 0, 4), Opt(OptOps.LOCAL, 0, 16), Opt(OptOps.UPCAST, 0, 2)]
|
||||||
|
opts += [Opt(OptOps.UPCAST, 1, 4), Opt(OptOps.LOCAL, 1, 16), Opt(OptOps.UPCAST, 1, 2)]
|
||||||
|
opts += [Opt(OptOps.UNROLL, 0, 8)]
|
||||||
|
|
||||||
|
return sink.replace(arg=KernelInfo(opts_to_apply=tuple(opts)))
|
||||||
|
|
||||||
def top_spec_kernel3():
|
def top_spec_kernel3():
|
||||||
a = Tensor.empty(N,N)
|
a = Tensor.empty(N,N)
|
||||||
b = Tensor.empty(N,N)
|
b = Tensor.empty(N,N)
|
||||||
c = a@b
|
c = a@b
|
||||||
sink = c.schedule()[-1].ast
|
sink = c.schedule()[-1].ast
|
||||||
L = 16
|
L = 16
|
||||||
sink = sink.reshape((N//L, L, N//L, L)) #.lift({0:UOp.range(dtypes.int, N//BM, 0), 2:UOp.range(dtypes.int, N//BN, 1)})
|
sink = sink.reshape((N//L, L, N//L, L)) #.lift({0:UOp.range(N//BM, 0), 2:UOp.range(N//BN, 1)})
|
||||||
sink = graph_rewrite(sink, view_left+pm)
|
sink = graph_rewrite(sink, view_left+pm)
|
||||||
axis_types = (AxisType.GLOBAL, AxisType.LOCAL, AxisType.GLOBAL, AxisType.LOCAL, AxisType.REDUCE)
|
axis_types = (AxisType.GLOBAL, AxisType.LOCAL, AxisType.GLOBAL, AxisType.LOCAL, AxisType.REDUCE)
|
||||||
return sink.replace(arg=KernelInfo(name="top_"+to_colored(sink.full_shape, axis_types), axis_types=axis_types))
|
return sink.replace(arg=KernelInfo(name="top_"+to_colored(sink.full_shape, axis_types), axis_types=axis_types))
|
||||||
@@ -171,7 +185,7 @@ def hand_spec_kernel3(kernel4=getenv("K4", 0), kernel5=getenv("K5", 0)):
|
|||||||
|
|
||||||
c_regs = UOp(Ops.DEFINE_REG, dtypes.float.ptr(TM * nbIterWaveM * TN * nbIterWaveN), arg=2)
|
c_regs = UOp(Ops.DEFINE_REG, dtypes.float.ptr(TM * nbIterWaveM * TN * nbIterWaveN), arg=2)
|
||||||
|
|
||||||
i = UOp.range(dtypes.int, c_regs.dtype.size, 16)
|
i = UOp.range(c_regs.dtype.size, 16)
|
||||||
init_store = c_regs[i].store(UOp.const(dtypes.float, 0.0), i)
|
init_store = c_regs[i].store(UOp.const(dtypes.float, 0.0), i)
|
||||||
|
|
||||||
if kernel4:
|
if kernel4:
|
||||||
@@ -182,53 +196,53 @@ def hand_spec_kernel3(kernel4=getenv("K4", 0), kernel5=getenv("K5", 0)):
|
|||||||
kId = 0
|
kId = 0
|
||||||
|
|
||||||
# load from globals into locals
|
# load from globals into locals
|
||||||
i = UOp.range(dtypes.int, nbReadsB, 0)
|
i = UOp.range(nbReadsB, 0)
|
||||||
index_x = BN * blockIdx_x + rBIdx
|
index_x = BN * blockIdx_x + rBIdx
|
||||||
index_y = rBIdy + i * strideReadB + kId
|
index_y = rBIdy + i * strideReadB + kId
|
||||||
Bs_store = Bs[(index_y % BK) * BN + index_x % BN].store(b[N * index_y + index_x].load(), i)
|
Bs_store = Bs[(index_y % BK) * BN + index_x % BN].store(b[N * index_y + index_x].load(), i)
|
||||||
|
|
||||||
i = UOp.range(dtypes.int, nbReadsA, 1)
|
i = UOp.range(nbReadsA, 1)
|
||||||
index_x = rAIdx + kId
|
index_x = rAIdx + kId
|
||||||
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
||||||
As_store = As[(index_x % BK) * BM_As_stride + index_y % BM].store(a[N * index_y + index_x].load(), i)
|
As_store = As[(index_x % BK) * BM_As_stride + index_y % BM].store(a[N * index_y + index_x].load(), i)
|
||||||
|
|
||||||
# iterate over the middle chunk
|
# iterate over the middle chunk
|
||||||
kId_range = UOp.range(dtypes.int, N//BK-1, 2)
|
kId_range = UOp.range(N//BK-1, 2)
|
||||||
kId = kId_range*BK
|
kId = kId_range*BK
|
||||||
|
|
||||||
barrier = UOp.barrier(As_store, Bs_store)
|
barrier = UOp.barrier(As_store, Bs_store)
|
||||||
|
|
||||||
# load from globals into registers (next round)
|
# load from globals into registers (next round)
|
||||||
i = UOp.range(dtypes.int, nbReadsB, 3)
|
i = UOp.range(nbReadsB, 3)
|
||||||
index_x = BN * blockIdx_x + rBIdx
|
index_x = BN * blockIdx_x + rBIdx
|
||||||
index_y = rBIdy + i * strideReadB + kId + BK
|
index_y = rBIdy + i * strideReadB + kId + BK
|
||||||
regB_store = regB[i].store(b[N * index_y + index_x].load(), i)
|
regB_store = regB[i].store(b[N * index_y + index_x].load(), i)
|
||||||
|
|
||||||
i = UOp.range(dtypes.int, nbReadsA, 4)
|
i = UOp.range(nbReadsA, 4)
|
||||||
index_x = rAIdx + kId + BK
|
index_x = rAIdx + kId + BK
|
||||||
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
||||||
regA_store = regA[i].store(a[N * index_y + index_x].load(), i)
|
regA_store = regA[i].store(a[N * index_y + index_x].load(), i)
|
||||||
|
|
||||||
def inner_loop(first_range, inp_dep=()):
|
def inner_loop(first_range, inp_dep=()):
|
||||||
# inner unroll
|
# inner unroll
|
||||||
k = UOp.range(dtypes.int, BK, first_range+0)
|
k = UOp.range(BK, first_range+0)
|
||||||
|
|
||||||
# load from locals into registers
|
# load from locals into registers
|
||||||
iterWave = UOp.range(dtypes.int, nbIterWaveN, first_range+1)
|
iterWave = UOp.range(nbIterWaveN, first_range+1)
|
||||||
i = UOp.range(dtypes.int, TN, first_range+2)
|
i = UOp.range(TN, first_range+2)
|
||||||
index = waveIdx * WN + iterWave * SUBWN + TN * idxInWave + i
|
index = waveIdx * WN + iterWave * SUBWN + TN * idxInWave + i
|
||||||
B_row_store = B_row[iterWave*TN + i].store(Bs[k*BN + index].load(*inp_dep), iterWave, i)
|
B_row_store = B_row[iterWave*TN + i].store(Bs[k*BN + index].load(*inp_dep), iterWave, i)
|
||||||
|
|
||||||
iterWave = UOp.range(dtypes.int, nbIterWaveM, first_range+3)
|
iterWave = UOp.range(nbIterWaveM, first_range+3)
|
||||||
i = UOp.range(dtypes.int, TM, first_range+4)
|
i = UOp.range(TM, first_range+4)
|
||||||
index = waveIdy * WM + iterWave * SUBWM + TM * idyInWave + i
|
index = waveIdy * WM + iterWave * SUBWM + TM * idyInWave + i
|
||||||
A_col_store = A_col[iterWave*TM + i].store(As[k*BM_As_stride + index].load(*inp_dep), iterWave, i)
|
A_col_store = A_col[iterWave*TM + i].store(As[k*BM_As_stride + index].load(*inp_dep), iterWave, i)
|
||||||
|
|
||||||
# do the GEMM math
|
# do the GEMM math
|
||||||
iterWaveM = UOp.range(dtypes.int, nbIterWaveM, first_range+5)
|
iterWaveM = UOp.range(nbIterWaveM, first_range+5)
|
||||||
yt = UOp.range(dtypes.int, TM, first_range+6)
|
yt = UOp.range(TM, first_range+6)
|
||||||
iterWaveN = UOp.range(dtypes.int, nbIterWaveN, first_range+7)
|
iterWaveN = UOp.range(nbIterWaveN, first_range+7)
|
||||||
xt = UOp.range(dtypes.int, TN, first_range+8)
|
xt = UOp.range(TN, first_range+8)
|
||||||
x = iterWaveN * TN + xt
|
x = iterWaveN * TN + xt
|
||||||
y = iterWaveM * TM + yt
|
y = iterWaveM * TM + yt
|
||||||
c_regs_idx = c_regs[y * TN * nbIterWaveN + x]
|
c_regs_idx = c_regs[y * TN * nbIterWaveN + x]
|
||||||
@@ -241,12 +255,12 @@ def hand_spec_kernel3(kernel4=getenv("K4", 0), kernel5=getenv("K5", 0)):
|
|||||||
sink = inner_loop(5, (barrier, regB_store, regA_store)).barrier()
|
sink = inner_loop(5, (barrier, regB_store, regA_store)).barrier()
|
||||||
|
|
||||||
# load from registers into locals
|
# load from registers into locals
|
||||||
i = UOp.range(dtypes.int, nbReadsB, 14)
|
i = UOp.range(nbReadsB, 14)
|
||||||
index_x = BN * blockIdx_x + rBIdx
|
index_x = BN * blockIdx_x + rBIdx
|
||||||
index_y = rBIdy + i * strideReadB + kId + BK
|
index_y = rBIdy + i * strideReadB + kId + BK
|
||||||
Bs_store = Bs[(index_y % BK) * BN + index_x % BN].store(regB[i].load(sink), i, kId_range)
|
Bs_store = Bs[(index_y % BK) * BN + index_x % BN].store(regB[i].load(sink), i, kId_range)
|
||||||
|
|
||||||
i = UOp.range(dtypes.int, nbReadsA, 15)
|
i = UOp.range(nbReadsA, 15)
|
||||||
index_x = rAIdx + kId + BK
|
index_x = rAIdx + kId + BK
|
||||||
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
||||||
As_store = As[(index_x % BK) * BM_As_stride + index_y % BM].store(regA[i].load(sink), i, kId_range)
|
As_store = As[(index_x % BK) * BM_As_stride + index_y % BM].store(regA[i].load(sink), i, kId_range)
|
||||||
@@ -254,40 +268,40 @@ def hand_spec_kernel3(kernel4=getenv("K4", 0), kernel5=getenv("K5", 0)):
|
|||||||
# final iteration without the copy
|
# final iteration without the copy
|
||||||
sink = inner_loop(16, (UOp.barrier(Bs_store, As_store),))
|
sink = inner_loop(16, (UOp.barrier(Bs_store, As_store),))
|
||||||
else:
|
else:
|
||||||
kId_range = UOp.range(dtypes.int, N//BK, 0)
|
kId_range = UOp.range(N//BK, 0)
|
||||||
kId = kId_range*BK
|
kId = kId_range*BK
|
||||||
|
|
||||||
# load from globals into locals
|
# load from globals into locals
|
||||||
i = UOp.range(dtypes.int, nbReadsB, 1)
|
i = UOp.range(nbReadsB, 1)
|
||||||
index_x = BN * blockIdx_x + rBIdx
|
index_x = BN * blockIdx_x + rBIdx
|
||||||
index_y = rBIdy + i * strideReadB + kId
|
index_y = rBIdy + i * strideReadB + kId
|
||||||
Bs_store = Bs[(index_y % BK) * BN + index_x % BN].store(b[N * index_y + index_x].load(), i)
|
Bs_store = Bs[(index_y % BK) * BN + index_x % BN].store(b[N * index_y + index_x].load(), i)
|
||||||
|
|
||||||
i = UOp.range(dtypes.int, nbReadsA, 2)
|
i = UOp.range(nbReadsA, 2)
|
||||||
index_x = rAIdx + kId
|
index_x = rAIdx + kId
|
||||||
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
index_y = BM * blockIdx_y + rAIdy + i * strideReadA
|
||||||
As_store = As[(index_x % BK) * BM_As_stride + index_y % BM].store(a[N * index_y + index_x].load(), i)
|
As_store = As[(index_x % BK) * BM_As_stride + index_y % BM].store(a[N * index_y + index_x].load(), i)
|
||||||
|
|
||||||
barrier = UOp.barrier(As_store, Bs_store)
|
barrier = UOp.barrier(As_store, Bs_store)
|
||||||
|
|
||||||
k = UOp.range(dtypes.int, BK, 3)
|
k = UOp.range(BK, 3)
|
||||||
|
|
||||||
# load from locals into registers
|
# load from locals into registers
|
||||||
iterWave = UOp.range(dtypes.int, nbIterWaveN, 4)
|
iterWave = UOp.range(nbIterWaveN, 4)
|
||||||
i = UOp.range(dtypes.int, TN, 5)
|
i = UOp.range(TN, 5)
|
||||||
index = waveIdx * WN + iterWave * SUBWN + TN * idxInWave + i
|
index = waveIdx * WN + iterWave * SUBWN + TN * idxInWave + i
|
||||||
B_row_store = B_row[iterWave*TN + i].store(Bs[k*BN + index].load(barrier), iterWave, i)
|
B_row_store = B_row[iterWave*TN + i].store(Bs[k*BN + index].load(barrier), iterWave, i)
|
||||||
|
|
||||||
iterWave = UOp.range(dtypes.int, nbIterWaveM, 6)
|
iterWave = UOp.range(nbIterWaveM, 6)
|
||||||
i = UOp.range(dtypes.int, TM, 7)
|
i = UOp.range(TM, 7)
|
||||||
index = waveIdy * WM + iterWave * SUBWM + TM * idyInWave + i
|
index = waveIdy * WM + iterWave * SUBWM + TM * idyInWave + i
|
||||||
A_col_store = A_col[iterWave*TM + i].store(As[k*BM_As_stride + index].load(barrier), iterWave, i)
|
A_col_store = A_col[iterWave*TM + i].store(As[k*BM_As_stride + index].load(barrier), iterWave, i)
|
||||||
|
|
||||||
# do the GEMM math
|
# do the GEMM math
|
||||||
iterWaveM = UOp.range(dtypes.int, nbIterWaveM, 8)
|
iterWaveM = UOp.range(nbIterWaveM, 8)
|
||||||
yt = UOp.range(dtypes.int, TM, 9)
|
yt = UOp.range(TM, 9)
|
||||||
iterWaveN = UOp.range(dtypes.int, nbIterWaveN, 10)
|
iterWaveN = UOp.range(nbIterWaveN, 10)
|
||||||
xt = UOp.range(dtypes.int, TN, 12)
|
xt = UOp.range(TN, 12)
|
||||||
x = iterWaveN * TN + xt
|
x = iterWaveN * TN + xt
|
||||||
y = iterWaveM * TM + yt
|
y = iterWaveM * TM + yt
|
||||||
c_regs_idx = c_regs[y * TN * nbIterWaveN + x]
|
c_regs_idx = c_regs[y * TN * nbIterWaveN + x]
|
||||||
@@ -295,10 +309,10 @@ def hand_spec_kernel3(kernel4=getenv("K4", 0), kernel5=getenv("K5", 0)):
|
|||||||
iterWaveM, iterWaveN, yt, xt, k, kId_range)
|
iterWaveM, iterWaveN, yt, xt, k, kId_range)
|
||||||
|
|
||||||
# store c_regs into c
|
# store c_regs into c
|
||||||
iterWaveM = UOp.range(dtypes.int, nbIterWaveM, 1000)
|
iterWaveM = UOp.range(nbIterWaveM, 1000)
|
||||||
yt = UOp.range(dtypes.int, TM, 1001)
|
yt = UOp.range(TM, 1001)
|
||||||
iterWaveN = UOp.range(dtypes.int, nbIterWaveN, 1002)
|
iterWaveN = UOp.range(nbIterWaveN, 1002)
|
||||||
xt = UOp.range(dtypes.int, TN, 1003)
|
xt = UOp.range(TN, 1003)
|
||||||
xOut = blockIdx_x * BN + waveIdx * WN + iterWaveN * SUBWN + TN * idxInWave
|
xOut = blockIdx_x * BN + waveIdx * WN + iterWaveN * SUBWN + TN * idxInWave
|
||||||
yOut = blockIdx_y * BM + waveIdy * WM + iterWaveM * SUBWM + TM * idyInWave
|
yOut = blockIdx_y * BM + waveIdy * WM + iterWaveM * SUBWM + TM * idyInWave
|
||||||
indexC = N * (yOut + yt) + xOut + xt
|
indexC = N * (yOut + yt) + xOut + xt
|
||||||
@@ -309,10 +323,15 @@ def hand_spec_kernel3(kernel4=getenv("K4", 0), kernel5=getenv("K5", 0)):
|
|||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
HL = getenv("HL")
|
HL = getenv("HL")
|
||||||
if HL == 2: hprg = top_spec_kernel3()
|
if HL == 3: hprg = rangeify_kernel3()
|
||||||
|
elif HL == 2: hprg = top_spec_kernel3()
|
||||||
elif HL == 1: hprg = hl_spec_kernel3()
|
elif HL == 1: hprg = hl_spec_kernel3()
|
||||||
else: hprg = hand_spec_kernel3()
|
else: hprg = hand_spec_kernel3()
|
||||||
prg = get_program(hprg, Device.default.renderer)
|
if HL == 3:
|
||||||
|
with Context(BLOCK_REORDER=0):
|
||||||
|
prg = get_program(hprg, Device.default.renderer)
|
||||||
|
else:
|
||||||
|
prg = get_program(hprg, Device.default.renderer)
|
||||||
print(prg.src)
|
print(prg.src)
|
||||||
if getenv("SRC"): exit(0)
|
if getenv("SRC"): exit(0)
|
||||||
hrunner = CompiledRunner(prg)
|
hrunner = CompiledRunner(prg)
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from tinygrad.runtime.ops_gpu import CLProgram, CLCompiler
|
from tinygrad.runtime.ops_cl import CLProgram, CLCompiler
|
||||||
from tinygrad import Device, dtypes
|
from tinygrad import Device, dtypes
|
||||||
from tinygrad.device import Buffer
|
from tinygrad.device import Buffer
|
||||||
from hexdump import hexdump
|
from hexdump import hexdump
|
||||||
@@ -11,7 +11,7 @@ from hexdump import hexdump
|
|||||||
# https://registry.khronos.org/OpenCL/extensions/intel/cl_intel_subgroup_split_matrix_multiply_accumulate.html
|
# https://registry.khronos.org/OpenCL/extensions/intel/cl_intel_subgroup_split_matrix_multiply_accumulate.html
|
||||||
# https://hc34.hotchips.org/assets/program/conference/day1/GPU%20HPC/Intel_s%20Ponte%20Vecchio%20GPU%20-%20Architecture%20Systems%20and%20Software%20FINAL.pdf
|
# https://hc34.hotchips.org/assets/program/conference/day1/GPU%20HPC/Intel_s%20Ponte%20Vecchio%20GPU%20-%20Architecture%20Systems%20and%20Software%20FINAL.pdf
|
||||||
|
|
||||||
device = Device["GPU"]
|
device = Device["CL"]
|
||||||
|
|
||||||
# NOTE: only the subgroup type 8 ones work
|
# NOTE: only the subgroup type 8 ones work
|
||||||
prog = CLProgram(device, "test", CLCompiler(device, "test").compile(f"""
|
prog = CLProgram(device, "test", CLCompiler(device, "test").compile(f"""
|
||||||
@@ -26,9 +26,9 @@ __kernel void test(__global float* data0, const __global int* data1, const __glo
|
|||||||
"""))
|
"""))
|
||||||
#with open("/tmp/test.elf", "wb") as f: f.write(prog.lib)
|
#with open("/tmp/test.elf", "wb") as f: f.write(prog.lib)
|
||||||
|
|
||||||
a = Buffer("GPU", 8, dtypes.float32).allocate()
|
a = Buffer("CL", 8, dtypes.float32).allocate()
|
||||||
b = Buffer("GPU", 0x10, dtypes.float16).allocate()
|
b = Buffer("CL", 0x10, dtypes.float16).allocate()
|
||||||
c = Buffer("GPU", 8*0x10, dtypes.float16).allocate()
|
c = Buffer("CL", 8*0x10, dtypes.float16).allocate()
|
||||||
|
|
||||||
row = np.array([1,2,3,4,5,6,7,8,1,2,3,4,5,6,7,8], np.float16)
|
row = np.array([1,2,3,4,5,6,7,8,1,2,3,4,5,6,7,8], np.float16)
|
||||||
mat = np.random.random((8, 0x10)).astype(np.float16)
|
mat = np.random.random((8, 0x10)).astype(np.float16)
|
||||||
|
|||||||
@@ -56,7 +56,7 @@ def randoms():
|
|||||||
def ast_to_cuda_prog(compiler, ast, opts):
|
def ast_to_cuda_prog(compiler, ast, opts):
|
||||||
k = Kernel(ast)
|
k = Kernel(ast)
|
||||||
k.apply_opts(opts)
|
k.apply_opts(opts)
|
||||||
p = get_program(k.get_optimized_ast(), k.opts)
|
p = get_program(k.ast, k.opts, k.applied_opts)
|
||||||
return CUDAProgram(device, p.function_name, compiler.compile(p.src))
|
return CUDAProgram(device, p.function_name, compiler.compile(p.src))
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
@@ -75,7 +75,7 @@ if __name__ == "__main__":
|
|||||||
|
|
||||||
if GEMM_VARIATION == "max" and (M%64)==0 and (N%128)==0 and (K%64)==0 and DTYPE_IN == dtypes.half and DTYPE_OUT == dtypes.float and DTYPE_ACC == dtypes.float:
|
if GEMM_VARIATION == "max" and (M%64)==0 and (N%128)==0 and (K%64)==0 and DTYPE_IN == dtypes.half and DTYPE_OUT == dtypes.float and DTYPE_ACC == dtypes.float:
|
||||||
print("Using CUDA and triton-generated kernel")
|
print("Using CUDA and triton-generated kernel")
|
||||||
# See nv_triton_gemm.annotated.ptx for PTX code which was generated from `PYTHONPATH=. DEBUG=6 CUDA=1 PTX=1 python3 extra/gemm/triton_nv_matmul.py`
|
# See nv_triton_gemm.annotated.ptx for PTX code which was generated from `PYTHONPATH=. DEBUG=6 CUDA=1 CUDA_PTX=1 python3 extra/gemm/triton_nv_matmul.py`
|
||||||
# this kernel with M=N=K=4096 does 162TFLOPS, vs torch at 144TFLOPS and BEAM=8 tinygrad at 138TFLOPS. theo max is 165TFLOPS.
|
# this kernel with M=N=K=4096 does 162TFLOPS, vs torch at 144TFLOPS and BEAM=8 tinygrad at 138TFLOPS. theo max is 165TFLOPS.
|
||||||
|
|
||||||
# WMMA element size is (M, N, K) = (16, 8, 16)
|
# WMMA element size is (M, N, K) = (16, 8, 16)
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ import numpy as np
|
|||||||
from tinygrad import dtypes, Tensor
|
from tinygrad import dtypes, Tensor
|
||||||
from tinygrad.helpers import getenv, get_single_element
|
from tinygrad.helpers import getenv, get_single_element
|
||||||
from tinygrad.dtype import _to_np_dtype
|
from tinygrad.dtype import _to_np_dtype
|
||||||
from tinygrad.codegen.opt.kernel import OptOps
|
from tinygrad.codegen.opt import OptOps
|
||||||
from tinygrad.engine.realize import lower_schedule
|
from tinygrad.engine.realize import lower_schedule
|
||||||
|
|
||||||
dtype_in = dtypes.half if getenv("HALF") else dtypes.bfloat16 if getenv("BFLOAT16") else dtypes.float
|
dtype_in = dtypes.half if getenv("HALF") else dtypes.bfloat16 if getenv("BFLOAT16") else dtypes.float
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ if __name__ == "__main__":
|
|||||||
Opt(op=OptOps.LOCAL, axis=0, amt=2),
|
Opt(op=OptOps.LOCAL, axis=0, amt=2),
|
||||||
]
|
]
|
||||||
k.apply_opts(opts)
|
k.apply_opts(opts)
|
||||||
prg = get_program(k.get_optimized_ast(), k.opts)
|
prg = get_program(k.ast, k.opts, k.applied_opts)
|
||||||
new_src = prg.src
|
new_src = prg.src
|
||||||
# can mod source here
|
# can mod source here
|
||||||
prg = replace(prg, src=new_src)
|
prg = replace(prg, src=new_src)
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ def matmul_kernel(c_ptr, a_ptr, b_ptr, BLOCK_SIZE_M: tl.constexpr, BLOCK_SIZE_N:
|
|||||||
c_ptrs = c_ptr + stride_cm * offs_cm[:, None] + stride_cn * offs_cn[None, :]
|
c_ptrs = c_ptr + stride_cm * offs_cm[:, None] + stride_cn * offs_cn[None, :]
|
||||||
tl.store(c_ptrs, c)
|
tl.store(c_ptrs, c)
|
||||||
|
|
||||||
# CUDA=1 PTX=1 python3 extra/gemm/triton_nv_matmul.py
|
# CUDA=1 CUDA_PTX=1 python3 extra/gemm/triton_nv_matmul.py
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
BLOCK_SIZE_M, BLOCK_SIZE_N, BLOCK_SIZE_K = 64, 128, 64
|
BLOCK_SIZE_M, BLOCK_SIZE_N, BLOCK_SIZE_K = 64, 128, 64
|
||||||
M, N, K = 4096, 4096, 4096
|
M, N, K = 4096, 4096, 4096
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ bert_train_params = {
|
|||||||
"GPUS": 6,
|
"GPUS": 6,
|
||||||
"BS": 96,
|
"BS": 96,
|
||||||
"EVAL_BS": 96,
|
"EVAL_BS": 96,
|
||||||
"FUSE_ARANGE": 1,
|
|
||||||
"BASEDIR": "/raid/datasets/wiki",
|
"BASEDIR": "/raid/datasets/wiki",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ def ioctls_from_header():
|
|||||||
hdr = (pathlib.Path(__file__).parent / "kfd_ioctl.h").read_text().replace("\\\n", "")
|
hdr = (pathlib.Path(__file__).parent / "kfd_ioctl.h").read_text().replace("\\\n", "")
|
||||||
pattern = r'#define\s+(AMDKFD_IOC_[A-Z0-9_]+)\s+AMDKFD_IOW?R?\((0x[0-9a-fA-F]+),\s+struct\s([A-Za-z0-9_]+)\)'
|
pattern = r'#define\s+(AMDKFD_IOC_[A-Z0-9_]+)\s+AMDKFD_IOW?R?\((0x[0-9a-fA-F]+),\s+struct\s([A-Za-z0-9_]+)\)'
|
||||||
matches = re.findall(pattern, hdr, re.MULTILINE)
|
matches = re.findall(pattern, hdr, re.MULTILINE)
|
||||||
return {int(nr, 0x10):(name, getattr(kfd_ioctl, "struct_"+sname)) for name, nr, sname in matches}
|
return {int(nr, 0x10):(name, getattr(kfd_ioctl, "struct_"+sname, None)) for name, nr, sname in matches}
|
||||||
nrs = ioctls_from_header()
|
nrs = ioctls_from_header()
|
||||||
|
|
||||||
@ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_int, ctypes.c_ulong, ctypes.c_void_p)
|
@ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_int, ctypes.c_ulong, ctypes.c_void_p)
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -1,7 +1,7 @@
|
|||||||
import onnx, yaml, tempfile, time, argparse, json
|
import onnx, yaml, tempfile, time, argparse, json
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
from tinygrad.frontend.onnx import OnnxRunner
|
from tinygrad.nn.onnx import OnnxRunner
|
||||||
from extra.onnx_helpers import validate, get_example_inputs
|
from extra.onnx_helpers import validate, get_example_inputs
|
||||||
from extra.huggingface_onnx.huggingface_manager import DOWNLOADS_DIR, snapshot_download_with_retry
|
from extra.huggingface_onnx.huggingface_manager import DOWNLOADS_DIR, snapshot_download_with_retry
|
||||||
|
|
||||||
|
|||||||
@@ -88,7 +88,7 @@ def mcts_search(lin:Kernel, rawbufs:List[Buffer], amt:int) -> Kernel:
|
|||||||
return ret
|
return ret
|
||||||
|
|
||||||
rawbufs = _ensure_buffer_alloc(rawbufs)
|
rawbufs = _ensure_buffer_alloc(rawbufs)
|
||||||
var_vals = {k:(k.vmax+k.vmin)//2 for k in lin.ast.variables()}
|
var_vals = {k.expr:(k.vmax+k.vmin)//2 for k in lin.ast.variables()}
|
||||||
dev = Device[lin.opts.device]
|
dev = Device[lin.opts.device]
|
||||||
root = MCTSNode(lin)
|
root = MCTSNode(lin)
|
||||||
|
|
||||||
|
|||||||
+32
-15
@@ -9,6 +9,9 @@ from PIL import Image
|
|||||||
import numpy as np
|
import numpy as np
|
||||||
import re, gzip
|
import re, gzip
|
||||||
|
|
||||||
|
# Allow for monkeypatching for mlperf.
|
||||||
|
gelu = Tensor.gelu
|
||||||
|
|
||||||
@lru_cache()
|
@lru_cache()
|
||||||
def default_bpe():
|
def default_bpe():
|
||||||
# Clip tokenizer, taken from https://github.com/openai/CLIP/blob/main/clip/simple_tokenizer.py (MIT license)
|
# Clip tokenizer, taken from https://github.com/openai/CLIP/blob/main/clip/simple_tokenizer.py (MIT license)
|
||||||
@@ -53,8 +56,8 @@ class Tokenizer:
|
|||||||
cs = [chr(n) for n in cs]
|
cs = [chr(n) for n in cs]
|
||||||
return dict(zip(bs, cs))
|
return dict(zip(bs, cs))
|
||||||
class ClipTokenizer:
|
class ClipTokenizer:
|
||||||
def __init__(self):
|
def __init__(self, version=None):
|
||||||
self.byte_encoder = Tokenizer.bytes_to_unicode()
|
self.byte_encoder, self.version = Tokenizer.bytes_to_unicode(), version
|
||||||
merges = gzip.open(default_bpe()).read().decode("utf-8").split('\n')
|
merges = gzip.open(default_bpe()).read().decode("utf-8").split('\n')
|
||||||
merges = merges[1:49152-256-2+1]
|
merges = merges[1:49152-256-2+1]
|
||||||
merges = [tuple(merge.split()) for merge in merges]
|
merges = [tuple(merge.split()) for merge in merges]
|
||||||
@@ -62,11 +65,17 @@ class Tokenizer:
|
|||||||
vocab = vocab + [v+'</w>' for v in vocab]
|
vocab = vocab + [v+'</w>' for v in vocab]
|
||||||
for merge in merges:
|
for merge in merges:
|
||||||
vocab.append(''.join(merge))
|
vocab.append(''.join(merge))
|
||||||
vocab.extend(['<|startoftext|>', '<|endoftext|>'])
|
if self.version == "sd_mlperf_v5_0":
|
||||||
|
import regex
|
||||||
|
vocab.extend(['<start_of_text>', '<end_of_text>'])
|
||||||
|
self.cache = {'<start_of_text>': '<start_of_text>', '<end_of_text>': '<end_of_text>'}
|
||||||
|
self.pat = regex.compile(r"""<start_of_text>|<end_of_text>|'s|'t|'re|'ve|'m|'ll|'d|[\p{L}]+|[\p{N}]|[^\s\p{L}\p{N}]+""", regex.IGNORECASE)
|
||||||
|
else:
|
||||||
|
vocab.extend(['<|startoftext|>', '<|endoftext|>'])
|
||||||
|
self.cache = {'<|startoftext|>': '<|startoftext|>', '<|endoftext|>': '<|endoftext|>'}
|
||||||
|
self.pat = re.compile(r"""<\|startoftext\|>|<\|endoftext\|>|'s|'t|'re|'ve|'m|'ll|'d|[^\s]+""", re.IGNORECASE)
|
||||||
self.encoder = dict(zip(vocab, range(len(vocab))))
|
self.encoder = dict(zip(vocab, range(len(vocab))))
|
||||||
self.bpe_ranks = dict(zip(merges, range(len(merges))))
|
self.bpe_ranks = dict(zip(merges, range(len(merges))))
|
||||||
self.cache = {'<|startoftext|>': '<|startoftext|>', '<|endoftext|>': '<|endoftext|>'}
|
|
||||||
self.pat = re.compile(r"""<\|startoftext\|>|<\|endoftext\|>|'s|'t|'re|'ve|'m|'ll|'d|[^\s]+""", re.IGNORECASE)
|
|
||||||
|
|
||||||
def bpe(self, token):
|
def bpe(self, token):
|
||||||
if token in self.cache:
|
if token in self.cache:
|
||||||
@@ -110,8 +119,17 @@ class Tokenizer:
|
|||||||
|
|
||||||
def encode(self, text:str, pad_with_zeros:bool=False) -> List[int]:
|
def encode(self, text:str, pad_with_zeros:bool=False) -> List[int]:
|
||||||
bpe_tokens: List[int] = []
|
bpe_tokens: List[int] = []
|
||||||
text = Tokenizer.whitespace_clean(text.strip()).lower()
|
if self.version == "sd_mlperf_v5_0":
|
||||||
for token in re.findall(self.pat, text):
|
import regex, ftfy, html
|
||||||
|
text = ftfy.fix_text(text)
|
||||||
|
text = html.unescape(html.unescape(text)).strip()
|
||||||
|
text = Tokenizer.whitespace_clean(text).lower()
|
||||||
|
re_module = regex
|
||||||
|
else:
|
||||||
|
text = Tokenizer.whitespace_clean(text.strip()).lower()
|
||||||
|
re_module = re
|
||||||
|
|
||||||
|
for token in re_module.findall(self.pat, text):
|
||||||
token = ''.join(self.byte_encoder[b] for b in token.encode('utf-8'))
|
token = ''.join(self.byte_encoder[b] for b in token.encode('utf-8'))
|
||||||
bpe_tokens.extend(self.encoder[bpe_token] for bpe_token in self.bpe(token).split(' '))
|
bpe_tokens.extend(self.encoder[bpe_token] for bpe_token in self.bpe(token).split(' '))
|
||||||
# Truncation, keeping two slots for start and end tokens.
|
# Truncation, keeping two slots for start and end tokens.
|
||||||
@@ -252,10 +270,8 @@ class Open:
|
|||||||
q,k,v = [y.reshape(T, B*self.n_heads, self.d_head).transpose(0, 1).reshape(B, self.n_heads, T, self.d_head) for y in proj.chunk(3)]
|
q,k,v = [y.reshape(T, B*self.n_heads, self.d_head).transpose(0, 1).reshape(B, self.n_heads, T, self.d_head) for y in proj.chunk(3)]
|
||||||
|
|
||||||
attn_output = Tensor.scaled_dot_product_attention(q, k, v, attn_mask=attn_mask)
|
attn_output = Tensor.scaled_dot_product_attention(q, k, v, attn_mask=attn_mask)
|
||||||
attn_output = attn_output.permute(2, 0, 1, 3).reshape(T*B, C)
|
attn_output = attn_output.permute(2, 0, 1, 3).reshape(T, B, C)
|
||||||
|
|
||||||
attn_output = self.out_proj(attn_output)
|
attn_output = self.out_proj(attn_output)
|
||||||
attn_output = attn_output.reshape(T, B, C)
|
|
||||||
|
|
||||||
return attn_output
|
return attn_output
|
||||||
|
|
||||||
@@ -263,9 +279,10 @@ class Open:
|
|||||||
def __init__(self, dims, hidden_dims):
|
def __init__(self, dims, hidden_dims):
|
||||||
self.c_fc = Linear(dims, hidden_dims)
|
self.c_fc = Linear(dims, hidden_dims)
|
||||||
self.c_proj = Linear(hidden_dims, dims)
|
self.c_proj = Linear(hidden_dims, dims)
|
||||||
|
self.gelu = gelu
|
||||||
|
|
||||||
def __call__(self, x:Tensor) -> Tensor:
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
return x.sequential([self.c_fc, Tensor.gelu, self.c_proj])
|
return x.sequential([self.c_fc, self.gelu, self.c_proj])
|
||||||
|
|
||||||
# https://github.com/mlfoundations/open_clip/blob/58e4e39aaabc6040839b0d2a7e8bf20979e4558a/src/open_clip/transformer.py#L210
|
# https://github.com/mlfoundations/open_clip/blob/58e4e39aaabc6040839b0d2a7e8bf20979e4558a/src/open_clip/transformer.py#L210
|
||||||
class ResidualAttentionBlock:
|
class ResidualAttentionBlock:
|
||||||
@@ -350,15 +367,15 @@ class Open:
|
|||||||
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/encoders/modules.py#L396
|
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/encoders/modules.py#L396
|
||||||
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/encoders/modules.py#L498
|
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/encoders/modules.py#L498
|
||||||
class FrozenOpenClipEmbedder(Embedder):
|
class FrozenOpenClipEmbedder(Embedder):
|
||||||
def __init__(self, dims:int, n_heads:int, layers:int, return_pooled:bool, ln_penultimate:bool=False):
|
def __init__(self, dims:int, n_heads:int, layers:int, return_pooled:bool, ln_penultimate:bool=False, clip_tokenizer_version=None):
|
||||||
self.tokenizer = Tokenizer.ClipTokenizer()
|
self.tokenizer = Tokenizer.ClipTokenizer(version=clip_tokenizer_version)
|
||||||
self.model = Open.ClipTextTransformer(dims, n_heads, layers)
|
self.model = Open.ClipTextTransformer(dims, n_heads, layers)
|
||||||
self.return_pooled = return_pooled
|
self.return_pooled = return_pooled
|
||||||
self.input_key = "txt"
|
self.input_key = "txt"
|
||||||
self.ln_penultimate = ln_penultimate
|
self.ln_penultimate = ln_penultimate
|
||||||
|
|
||||||
def tokenize(self, text:str, device:Optional[str]=None) -> Tensor:
|
def tokenize(self, text:str, device:Optional[str]=None) -> Tensor:
|
||||||
return Tensor(self.tokenizer.encode(text, pad_with_zeros=True), dtype=dtypes.int64, device=device).reshape(1,-1)
|
return Tensor(self.tokenizer.encode(text, pad_with_zeros=True), dtype=dtypes.int32, device=device).reshape(1,-1)
|
||||||
|
|
||||||
def text_transformer_forward(self, x:Tensor, attn_mask:Optional[Tensor]=None):
|
def text_transformer_forward(self, x:Tensor, attn_mask:Optional[Tensor]=None):
|
||||||
for r in self.model.transformer.resblocks:
|
for r in self.model.transformer.resblocks:
|
||||||
@@ -449,7 +466,7 @@ class OpenClipEncoder:
|
|||||||
x = x + self.positional_embedding
|
x = x + self.positional_embedding
|
||||||
x = self.transformer(x, attn_mask=self.attn_mask)
|
x = self.transformer(x, attn_mask=self.attn_mask)
|
||||||
x = self.ln_final(x)
|
x = self.ln_final(x)
|
||||||
x = x[:, tokens.argmax(axis=-1)]
|
x = x[Tensor.arange(x.shape[0], device=x.device), tokens.argmax(axis=-1)]
|
||||||
x = x @ self.text_projection
|
x = x @ self.text_projection
|
||||||
return x
|
return x
|
||||||
|
|
||||||
|
|||||||
@@ -270,8 +270,10 @@ class FidInceptionV3:
|
|||||||
self.Mixed_7b = inception.Mixed_7b
|
self.Mixed_7b = inception.Mixed_7b
|
||||||
self.Mixed_7c = inception.Mixed_7c
|
self.Mixed_7c = inception.Mixed_7c
|
||||||
|
|
||||||
def load_from_pretrained(self):
|
def load_from_pretrained(self, path=None):
|
||||||
state_dict = torch_load(str(fetch("https://github.com/mseitzer/pytorch-fid/releases/download/fid_weights/pt_inception-2015-12-05-6726825d.pth", "pt_inception-2015-12-05-6726825d.pth")))
|
if path is None:
|
||||||
|
path = fetch("https://github.com/mseitzer/pytorch-fid/releases/download/fid_weights/pt_inception-2015-12-05-6726825d.pth", "pt_inception-2015-12-05-6726825d.pth")
|
||||||
|
state_dict = torch_load(str(path))
|
||||||
for k,v in state_dict.items():
|
for k,v in state_dict.items():
|
||||||
if k.endswith(".num_batches_tracked"):
|
if k.endswith(".num_batches_tracked"):
|
||||||
state_dict[k] = v.reshape(1)
|
state_dict[k] = v.reshape(1)
|
||||||
|
|||||||
@@ -249,8 +249,5 @@ def convert_from_gguf(weights:dict[str, Tensor], n_layers:int):
|
|||||||
return sd
|
return sd
|
||||||
|
|
||||||
def fix_bf16(weights:dict[Any, Tensor]):
|
def fix_bf16(weights:dict[Any, Tensor]):
|
||||||
if getenv("SUPPORT_BF16", 1):
|
# TODO: without casting to float16, 70B llama OOM on tinybox.
|
||||||
# TODO: without casting to float16, 70B llama OOM on tinybox.
|
return {k:v.cast(dtypes.float32).cast(dtypes.float16) if v.dtype == dtypes.bfloat16 else v for k,v in weights.items()}
|
||||||
return {k:v.cast(dtypes.float32).cast(dtypes.float16) if v.dtype == dtypes.bfloat16 else v for k,v in weights.items()}
|
|
||||||
# TODO: check if device supports bf16
|
|
||||||
return {k:v.llvm_bf16_cast(dtypes.half).to(v.device) if v.dtype == dtypes.bfloat16 else v for k,v in weights.items()}
|
|
||||||
|
|||||||
+35
-27
@@ -1,21 +1,24 @@
|
|||||||
from tinygrad import Tensor, dtypes
|
from tinygrad import Tensor, dtypes, nn
|
||||||
from tinygrad.nn import Linear, Conv2d, GroupNorm, LayerNorm
|
|
||||||
from tinygrad.device import is_dtype_supported
|
from tinygrad.device import is_dtype_supported
|
||||||
from typing import Optional, Union, List, Any, Tuple
|
from typing import Optional, Union, List, Any, Tuple, Callable
|
||||||
import math
|
import math
|
||||||
|
|
||||||
|
# allow for monkeypatching
|
||||||
|
Linear, Conv2d, GroupNorm, LayerNorm = nn.Linear, nn.Conv2d, nn.GroupNorm, nn.LayerNorm
|
||||||
|
attention, gelu, mixed_precision_dtype = Tensor.scaled_dot_product_attention, Tensor.gelu, dtypes.float16
|
||||||
|
|
||||||
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/diffusionmodules/util.py#L207
|
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/diffusionmodules/util.py#L207
|
||||||
def timestep_embedding(timesteps:Tensor, dim:int, max_period=10000):
|
def timestep_embedding(timesteps:Tensor, dim:int, max_period=10000):
|
||||||
half = dim // 2
|
half = dim // 2
|
||||||
freqs = (-math.log(max_period) * Tensor.arange(half, device=timesteps.device) / half).exp()
|
freqs = (-math.log(max_period) * Tensor.arange(half, device=timesteps.device) / half).exp()
|
||||||
args = timesteps.unsqueeze(1) * freqs.unsqueeze(0)
|
args = timesteps.unsqueeze(1) * freqs.unsqueeze(0)
|
||||||
out = Tensor.cat(args.cos(), args.sin(), dim=-1)
|
out = Tensor.cat(args.cos(), args.sin(), dim=-1)
|
||||||
return out.cast(dtypes.float16) if is_dtype_supported(dtypes.float16) else out
|
return out.cast(mixed_precision_dtype) if is_dtype_supported(mixed_precision_dtype) else out
|
||||||
|
|
||||||
class ResBlock:
|
class ResBlock:
|
||||||
def __init__(self, channels:int, emb_channels:int, out_channels:int):
|
def __init__(self, channels:int, emb_channels:int, out_channels:int, num_groups:int=32):
|
||||||
self.in_layers = [
|
self.in_layers = [
|
||||||
GroupNorm(32, channels),
|
GroupNorm(num_groups, channels),
|
||||||
Tensor.silu,
|
Tensor.silu,
|
||||||
Conv2d(channels, out_channels, 3, padding=1),
|
Conv2d(channels, out_channels, 3, padding=1),
|
||||||
]
|
]
|
||||||
@@ -24,7 +27,7 @@ class ResBlock:
|
|||||||
Linear(emb_channels, out_channels),
|
Linear(emb_channels, out_channels),
|
||||||
]
|
]
|
||||||
self.out_layers = [
|
self.out_layers = [
|
||||||
GroupNorm(32, out_channels),
|
GroupNorm(num_groups, out_channels),
|
||||||
Tensor.silu,
|
Tensor.silu,
|
||||||
lambda x: x, # needed for weights loading code to work
|
lambda x: x, # needed for weights loading code to work
|
||||||
Conv2d(out_channels, out_channels, 3, padding=1),
|
Conv2d(out_channels, out_channels, 3, padding=1),
|
||||||
@@ -45,35 +48,37 @@ class CrossAttention:
|
|||||||
self.to_v = Linear(ctx_dim, n_heads*d_head, bias=False)
|
self.to_v = Linear(ctx_dim, n_heads*d_head, bias=False)
|
||||||
self.num_heads = n_heads
|
self.num_heads = n_heads
|
||||||
self.head_size = d_head
|
self.head_size = d_head
|
||||||
|
self.attn = attention
|
||||||
self.to_out = [Linear(n_heads*d_head, query_dim)]
|
self.to_out = [Linear(n_heads*d_head, query_dim)]
|
||||||
|
|
||||||
def __call__(self, x:Tensor, ctx:Optional[Tensor]=None) -> Tensor:
|
def __call__(self, x:Tensor, ctx:Optional[Tensor]=None) -> Tensor:
|
||||||
ctx = x if ctx is None else ctx
|
ctx = x if ctx is None else ctx
|
||||||
q,k,v = self.to_q(x), self.to_k(ctx), self.to_v(ctx)
|
q,k,v = self.to_q(x), self.to_k(ctx), self.to_v(ctx)
|
||||||
q,k,v = [y.reshape(x.shape[0], -1, self.num_heads, self.head_size).transpose(1,2) for y in (q,k,v)]
|
q,k,v = [y.reshape(x.shape[0], -1, self.num_heads, self.head_size).transpose(1,2) for y in (q,k,v)]
|
||||||
attention = Tensor.scaled_dot_product_attention(q, k, v).transpose(1,2)
|
attention = self.attn(q, k, v).transpose(1,2)
|
||||||
h_ = attention.reshape(x.shape[0], -1, self.num_heads * self.head_size)
|
h_ = attention.reshape(x.shape[0], -1, self.num_heads * self.head_size)
|
||||||
return h_.sequential(self.to_out)
|
return h_.sequential(self.to_out)
|
||||||
|
|
||||||
class GEGLU:
|
class GEGLU:
|
||||||
def __init__(self, dim_in:int, dim_out:int):
|
def __init__(self, dim_in:int, dim_out:int):
|
||||||
self.proj = Linear(dim_in, dim_out * 2)
|
self.proj = Linear(dim_in, dim_out * 2)
|
||||||
|
self.gelu = gelu
|
||||||
self.dim_out = dim_out
|
self.dim_out = dim_out
|
||||||
|
|
||||||
def __call__(self, x:Tensor) -> Tensor:
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
x, gate = self.proj(x).chunk(2, dim=-1)
|
x, gate = self.proj(x).chunk(2, dim=-1)
|
||||||
return x * gate.gelu()
|
return x * self.gelu(gate)
|
||||||
|
|
||||||
class FeedForward:
|
class FeedForward:
|
||||||
def __init__(self, dim:int, mult:int=4):
|
def __init__(self, dim:int, mult:int=4):
|
||||||
self.net = [
|
self.net: tuple[GEGLU, Callable, nn.Linear] = (
|
||||||
GEGLU(dim, dim*mult),
|
GEGLU(dim, dim*mult),
|
||||||
lambda x: x, # needed for weights loading code to work
|
lambda x: x, # needed for weights loading code to work
|
||||||
Linear(dim*mult, dim)
|
Linear(dim*mult, dim)
|
||||||
]
|
)
|
||||||
|
|
||||||
def __call__(self, x:Tensor) -> Tensor:
|
def __call__(self, x:Tensor) -> Tensor:
|
||||||
return x.sequential(self.net)
|
return x.sequential(list(self.net))
|
||||||
|
|
||||||
class BasicTransformerBlock:
|
class BasicTransformerBlock:
|
||||||
def __init__(self, dim:int, ctx_dim:int, n_heads:int, d_head:int):
|
def __init__(self, dim:int, ctx_dim:int, n_heads:int, d_head:int):
|
||||||
@@ -92,12 +97,13 @@ class BasicTransformerBlock:
|
|||||||
|
|
||||||
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/attention.py#L619
|
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/attention.py#L619
|
||||||
class SpatialTransformer:
|
class SpatialTransformer:
|
||||||
def __init__(self, channels:int, n_heads:int, d_head:int, ctx_dim:Union[int,List[int]], use_linear:bool, depth:int=1):
|
def __init__(self, channels:int, n_heads:int, d_head:int, ctx_dim:Union[int,List[int]], use_linear:bool, depth:int=1,
|
||||||
|
norm_eps:float=1e-5):
|
||||||
if isinstance(ctx_dim, int):
|
if isinstance(ctx_dim, int):
|
||||||
ctx_dim = [ctx_dim]*depth
|
ctx_dim = [ctx_dim]*depth
|
||||||
else:
|
else:
|
||||||
assert isinstance(ctx_dim, list) and depth == len(ctx_dim)
|
assert isinstance(ctx_dim, list) and depth == len(ctx_dim)
|
||||||
self.norm = GroupNorm(32, channels)
|
self.norm = GroupNorm(32, channels, eps=norm_eps)
|
||||||
assert channels == n_heads * d_head
|
assert channels == n_heads * d_head
|
||||||
self.proj_in = Linear(channels, channels) if use_linear else Conv2d(channels, channels, 1)
|
self.proj_in = Linear(channels, channels) if use_linear else Conv2d(channels, channels, 1)
|
||||||
self.transformer_blocks = [BasicTransformerBlock(channels, ctx_dim[d], n_heads, d_head) for d in range(depth)]
|
self.transformer_blocks = [BasicTransformerBlock(channels, ctx_dim[d], n_heads, d_head) for d in range(depth)]
|
||||||
@@ -134,7 +140,9 @@ class Upsample:
|
|||||||
|
|
||||||
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/diffusionmodules/openaimodel.py#L472
|
# https://github.com/Stability-AI/generative-models/blob/fbdc58cab9f4ee2be7a5e1f2e2787ecd9311942f/sgm/modules/diffusionmodules/openaimodel.py#L472
|
||||||
class UNetModel:
|
class UNetModel:
|
||||||
def __init__(self, adm_in_ch:Optional[int], in_ch:int, out_ch:int, model_ch:int, attention_resolutions:List[int], num_res_blocks:int, channel_mult:List[int], transformer_depth:List[int], ctx_dim:Union[int,List[int]], use_linear:bool=False, d_head:Optional[int]=None, n_heads:Optional[int]=None):
|
def __init__(self, adm_in_ch:Optional[int], in_ch:int, out_ch:int, model_ch:int, attention_resolutions:List[int], num_res_blocks:int,
|
||||||
|
channel_mult:List[int], transformer_depth:List[int], ctx_dim:Union[int,List[int]], use_linear:bool=False, d_head:Optional[int]=None,
|
||||||
|
n_heads:Optional[int]=None, num_groups:int=32, st_norm_eps:float=1e-5):
|
||||||
self.model_ch = model_ch
|
self.model_ch = model_ch
|
||||||
self.num_res_blocks = [num_res_blocks] * len(channel_mult)
|
self.num_res_blocks = [num_res_blocks] * len(channel_mult)
|
||||||
|
|
||||||
@@ -174,12 +182,12 @@ class UNetModel:
|
|||||||
for idx, mult in enumerate(channel_mult):
|
for idx, mult in enumerate(channel_mult):
|
||||||
for _ in range(self.num_res_blocks[idx]):
|
for _ in range(self.num_res_blocks[idx]):
|
||||||
layers: List[Any] = [
|
layers: List[Any] = [
|
||||||
ResBlock(ch, time_embed_dim, model_ch*mult),
|
ResBlock(ch, time_embed_dim, model_ch*mult, num_groups),
|
||||||
]
|
]
|
||||||
ch = mult * model_ch
|
ch = mult * model_ch
|
||||||
if ds in attention_resolutions:
|
if ds in attention_resolutions:
|
||||||
d_head, n_heads = get_d_and_n_heads(ch)
|
d_head, n_heads = get_d_and_n_heads(ch)
|
||||||
layers.append(SpatialTransformer(ch, n_heads, d_head, ctx_dim, use_linear, depth=transformer_depth[idx]))
|
layers.append(SpatialTransformer(ch, n_heads, d_head, ctx_dim, use_linear, depth=transformer_depth[idx], norm_eps=st_norm_eps))
|
||||||
|
|
||||||
self.input_blocks.append(layers)
|
self.input_blocks.append(layers)
|
||||||
input_block_channels.append(ch)
|
input_block_channels.append(ch)
|
||||||
@@ -193,9 +201,9 @@ class UNetModel:
|
|||||||
|
|
||||||
d_head, n_heads = get_d_and_n_heads(ch)
|
d_head, n_heads = get_d_and_n_heads(ch)
|
||||||
self.middle_block: List = [
|
self.middle_block: List = [
|
||||||
ResBlock(ch, time_embed_dim, ch),
|
ResBlock(ch, time_embed_dim, ch, num_groups),
|
||||||
SpatialTransformer(ch, n_heads, d_head, ctx_dim, use_linear, depth=transformer_depth[-1]),
|
SpatialTransformer(ch, n_heads, d_head, ctx_dim, use_linear, depth=transformer_depth[-1], norm_eps=st_norm_eps),
|
||||||
ResBlock(ch, time_embed_dim, ch),
|
ResBlock(ch, time_embed_dim, ch, num_groups),
|
||||||
]
|
]
|
||||||
|
|
||||||
self.output_blocks = []
|
self.output_blocks = []
|
||||||
@@ -203,13 +211,13 @@ class UNetModel:
|
|||||||
for i in range(self.num_res_blocks[idx] + 1):
|
for i in range(self.num_res_blocks[idx] + 1):
|
||||||
ich = input_block_channels.pop()
|
ich = input_block_channels.pop()
|
||||||
layers = [
|
layers = [
|
||||||
ResBlock(ch + ich, time_embed_dim, model_ch*mult),
|
ResBlock(ch + ich, time_embed_dim, model_ch*mult, num_groups),
|
||||||
]
|
]
|
||||||
ch = model_ch * mult
|
ch = model_ch * mult
|
||||||
|
|
||||||
if ds in attention_resolutions:
|
if ds in attention_resolutions:
|
||||||
d_head, n_heads = get_d_and_n_heads(ch)
|
d_head, n_heads = get_d_and_n_heads(ch)
|
||||||
layers.append(SpatialTransformer(ch, n_heads, d_head, ctx_dim, use_linear, depth=transformer_depth[idx]))
|
layers.append(SpatialTransformer(ch, n_heads, d_head, ctx_dim, use_linear, depth=transformer_depth[idx], norm_eps=st_norm_eps))
|
||||||
|
|
||||||
if idx > 0 and i == self.num_res_blocks[idx]:
|
if idx > 0 and i == self.num_res_blocks[idx]:
|
||||||
layers.append(Upsample(ch))
|
layers.append(Upsample(ch))
|
||||||
@@ -217,7 +225,7 @@ class UNetModel:
|
|||||||
self.output_blocks.append(layers)
|
self.output_blocks.append(layers)
|
||||||
|
|
||||||
self.out = [
|
self.out = [
|
||||||
GroupNorm(32, ch),
|
GroupNorm(num_groups, ch),
|
||||||
Tensor.silu,
|
Tensor.silu,
|
||||||
Conv2d(model_ch, out_ch, 3, padding=1),
|
Conv2d(model_ch, out_ch, 3, padding=1),
|
||||||
]
|
]
|
||||||
@@ -230,10 +238,10 @@ class UNetModel:
|
|||||||
assert y.shape[0] == x.shape[0]
|
assert y.shape[0] == x.shape[0]
|
||||||
emb = emb + y.sequential(self.label_emb[0])
|
emb = emb + y.sequential(self.label_emb[0])
|
||||||
|
|
||||||
if is_dtype_supported(dtypes.float16):
|
if is_dtype_supported(mixed_precision_dtype):
|
||||||
emb = emb.cast(dtypes.float16)
|
emb = emb.cast(mixed_precision_dtype)
|
||||||
ctx = ctx.cast(dtypes.float16)
|
ctx = ctx.cast(mixed_precision_dtype)
|
||||||
x = x .cast(dtypes.float16)
|
x = x .cast(mixed_precision_dtype)
|
||||||
|
|
||||||
def run(x:Tensor, bb) -> Tensor:
|
def run(x:Tensor, bb) -> Tensor:
|
||||||
if isinstance(bb, ResBlock): x = bb(x, emb)
|
if isinstance(bb, ResBlock): x = bb(x, emb)
|
||||||
|
|||||||
@@ -272,4 +272,4 @@ def compare_launch_state(states, good_states):
|
|||||||
|
|
||||||
return True, "PASS"
|
return True, "PASS"
|
||||||
|
|
||||||
# IOCTL=1 PTX=1 CUDA=1 python3 test/test_ops.py TestOps.test_tiny_add
|
# IOCTL=1 CUDA=1 CUDA_PTX=1 python3 test/test_ops.py TestOps.test_tiny_add
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
from tinygrad import Tensor
|
from tinygrad import Tensor
|
||||||
from tinygrad.tensor import _to_np_dtype
|
from tinygrad.tensor import _to_np_dtype
|
||||||
from tinygrad.frontend.onnx import OnnxRunner, OnnxValue
|
from tinygrad.nn.onnx import OnnxRunner, OnnxValue
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import onnxruntime as ort
|
import onnxruntime as ort
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ rm $LOGOPS
|
|||||||
test/external/process_replay/reset.py
|
test/external/process_replay/reset.py
|
||||||
|
|
||||||
CI=1 python3 -m pytest -n=auto test/test_ops.py test/test_nn.py test/test_winograd.py test/models/test_real_world.py --durations=20
|
CI=1 python3 -m pytest -n=auto test/test_ops.py test/test_nn.py test/test_winograd.py test/models/test_real_world.py --durations=20
|
||||||
GPU=1 python3 -m pytest test/test_tiny.py
|
CL=1 python3 -m pytest test/test_tiny.py
|
||||||
|
|
||||||
# extract, sort and uniq
|
# extract, sort and uniq
|
||||||
extra/optimization/extract_dataset.py
|
extra/optimization/extract_dataset.py
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# stuff needed to unpack a kernel
|
# stuff needed to unpack a kernel
|
||||||
from tinygrad import Variable
|
from tinygrad import Variable
|
||||||
from tinygrad.codegen.opt.kernel import Opt, OptOps
|
from tinygrad.codegen.opt import Opt, OptOps
|
||||||
from tinygrad.uop.ops import UOp, Ops, KernelInfo
|
from tinygrad.uop.ops import UOp, Ops, KernelInfo
|
||||||
from tinygrad.dtype import dtypes, PtrDType
|
from tinygrad.dtype import dtypes, PtrDType
|
||||||
from tinygrad.shape.shapetracker import ShapeTracker
|
from tinygrad.shape.shapetracker import ShapeTracker
|
||||||
@@ -81,7 +81,7 @@ def lin_to_feats(lin:Kernel, use_sts=True):
|
|||||||
ret = [float(x) for x in ret]
|
ret = [float(x) for x in ret]
|
||||||
|
|
||||||
if use_sts:
|
if use_sts:
|
||||||
my_sts = dedup([(x.shape == lin.full_shape, x.real_strides(), any(v.mask is not None for v in x.views), len(x.views)) for x in lin.sts])
|
my_sts = dedup([(x.shape == lin.full_shape, x.is_expanded(), any(v.mask is not None for v in x.views), len(x.views)) for x in lin.sts])
|
||||||
assert len(my_sts) < MAX_BUFS
|
assert len(my_sts) < MAX_BUFS
|
||||||
sts_len = 3 + 5*MAX_DIMS
|
sts_len = 3 + 5*MAX_DIMS
|
||||||
for s in my_sts:
|
for s in my_sts:
|
||||||
@@ -115,7 +115,7 @@ def time_linearizer(lin:Kernel, rawbufs:list[Buffer], allow_test_size=True, max_
|
|||||||
assert dev.compiler is not None
|
assert dev.compiler is not None
|
||||||
|
|
||||||
rawbufs = _ensure_buffer_alloc(rawbufs)
|
rawbufs = _ensure_buffer_alloc(rawbufs)
|
||||||
var_vals: dict[Variable, int] = {k:int(k.vmax+k.vmin)//2 for k in lin.ast.variables()}
|
var_vals: dict[str, int] = {k.expr:int(k.vmax+k.vmin)//2 for k in lin.ast.variables()}
|
||||||
p = get_program(lin.get_optimized_ast(), lin.opts)
|
p = get_program(lin.get_optimized_ast(), lin.opts)
|
||||||
tms = _time_program(p, dev.compiler.compile(p.src), var_vals, rawbufs,
|
tms = _time_program(p, dev.compiler.compile(p.src), var_vals, rawbufs,
|
||||||
max_global_size=max_global_size if allow_test_size else None, clear_l2=clear_l2, cnt=cnt, name=to_function_name(lin.name))
|
max_global_size=max_global_size if allow_test_size else None, clear_l2=clear_l2, cnt=cnt, name=to_function_name(lin.name))
|
||||||
|
|||||||
@@ -16,9 +16,9 @@ class TestBeamSearch(unittest.TestCase):
|
|||||||
BEAM.value = self.old_beam
|
BEAM.value = self.old_beam
|
||||||
|
|
||||||
def test_variable_ast_beam(self):
|
def test_variable_ast_beam(self):
|
||||||
with Context(IGNORE_OOB=1):
|
vi = Variable("a", 1, 10).bind(3)
|
||||||
a = rand(3, 3).reshape((Variable("a", 1, 10).bind(3), 3))
|
a = rand(10, 3)[:vi]
|
||||||
a = (a+1).realize()
|
a = (a+1).realize()
|
||||||
|
|
||||||
def test_big_prime_number(self):
|
def test_big_prime_number(self):
|
||||||
a = rand(367, 367)
|
a = rand(367, 367)
|
||||||
@@ -42,18 +42,16 @@ class TestBeamSearch(unittest.TestCase):
|
|||||||
|
|
||||||
def test_variable_big_prime_number(self):
|
def test_variable_big_prime_number(self):
|
||||||
v = Variable("v", 1, 400).bind(367)
|
v = Variable("v", 1, 400).bind(367)
|
||||||
a = rand(367, 367)
|
a = rand(367, 400)
|
||||||
b = rand(367, 367)
|
b = rand(400, 367)
|
||||||
with Context(IGNORE_OOB=1):
|
c = (a[:, :v] @ b[:v, :]).realize()
|
||||||
c = (a.reshape(367, v) @ b.reshape(v, 367)).realize()
|
np.testing.assert_allclose(c.numpy(), a[:, :367].numpy() @ b[:367, :].numpy(), atol=1e-4, rtol=1e-4)
|
||||||
np.testing.assert_allclose(c.numpy(), a.numpy() @ b.numpy(), atol=1e-4, rtol=1e-4)
|
|
||||||
|
|
||||||
def test_variable_shrink_prime_number(self):
|
def test_variable_shrink_prime_number(self):
|
||||||
v = Variable("v", 1, 400).bind(367)
|
v = Variable("v", 1, 400).bind(367)
|
||||||
a = rand(400, 367)
|
a = rand(400, 367)
|
||||||
with Context(IGNORE_OOB=1):
|
b = (a.shrink(((0,v), None))+1)[:367,:367].realize()
|
||||||
b = (a.shrink(((0,v), None))+1).reshape(367,367).realize()
|
np.testing.assert_allclose(b.numpy(), a.numpy()[:367]+1, atol=1e-4, rtol=1e-4)
|
||||||
np.testing.assert_allclose(b.numpy(), a.numpy()[:367]+1, atol=1e-4, rtol=1e-4)
|
|
||||||
|
|
||||||
def test_no_mutate_rawbuffers(self):
|
def test_no_mutate_rawbuffers(self):
|
||||||
a = rand(3, 3).realize()
|
a = rand(3, 3).realize()
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
import ctypes, array
|
import ctypes, array
|
||||||
from hexdump import hexdump
|
from hexdump import hexdump
|
||||||
from tinygrad.runtime.ops_gpu import GPUDevice
|
from tinygrad.runtime.ops_cl import CLDevice
|
||||||
from tinygrad.helpers import getenv, to_mv, mv_address
|
from tinygrad.helpers import getenv, to_mv, mv_address
|
||||||
from tinygrad.dtype import dtypes
|
from tinygrad.dtype import dtypes
|
||||||
from tinygrad import Tensor, TinyJit
|
from tinygrad import Tensor, TinyJit
|
||||||
@@ -8,7 +8,7 @@ from tinygrad.runtime.autogen import opencl as cl
|
|||||||
if getenv("IOCTL"): import extra.qcom_gpu_driver.opencl_ioctl # noqa: F401 # pylint: disable=unused-import
|
if getenv("IOCTL"): import extra.qcom_gpu_driver.opencl_ioctl # noqa: F401 # pylint: disable=unused-import
|
||||||
|
|
||||||
# create raw opencl buffer.
|
# create raw opencl buffer.
|
||||||
gdev = GPUDevice()
|
gdev = CLDevice()
|
||||||
cl_buf = cl.clCreateBuffer(gdev.context, cl.CL_MEM_READ_WRITE, 0x100, None, status := ctypes.c_int32())
|
cl_buf = cl.clCreateBuffer(gdev.context, cl.CL_MEM_READ_WRITE, 0x100, None, status := ctypes.c_int32())
|
||||||
assert status.value == 0
|
assert status.value == 0
|
||||||
|
|
||||||
|
|||||||
@@ -673,6 +673,7 @@ impl<'a> Thread<'a> {
|
|||||||
39 => f32::log2(s0),
|
39 => f32::log2(s0),
|
||||||
42 => 1.0 / s0,
|
42 => 1.0 / s0,
|
||||||
43 => 1.0 / s0,
|
43 => 1.0 / s0,
|
||||||
|
46 => 1.0 / f32::sqrt(s0),
|
||||||
51 => f32::sqrt(s0),
|
51 => f32::sqrt(s0),
|
||||||
_ => todo_instr!(instruction)?,
|
_ => todo_instr!(instruction)?,
|
||||||
}
|
}
|
||||||
@@ -929,7 +930,7 @@ impl<'a> Thread<'a> {
|
|||||||
|
|
||||||
let op = ((instr >> 16) & 0x3ff) as u32;
|
let op = ((instr >> 16) & 0x3ff) as u32;
|
||||||
match op {
|
match op {
|
||||||
764 | 765 | 288 | 289 | 290 | 766 | 768 | 769 => {
|
764 | 765 | 288 | 289 | 290 | 766 | 767 | 768 | 769 => {
|
||||||
let vdst = (instr & 0xff) as usize;
|
let vdst = (instr & 0xff) as usize;
|
||||||
let sdst = ((instr >> 8) & 0x7f) as usize;
|
let sdst = ((instr >> 8) & 0x7f) as usize;
|
||||||
let f = |i: u32| -> usize { ((instr >> i) & 0x1ff) as usize };
|
let f = |i: u32| -> usize { ((instr >> i) & 0x1ff) as usize };
|
||||||
@@ -943,6 +944,16 @@ impl<'a> Thread<'a> {
|
|||||||
assert_eq!(clmp, 0);
|
assert_eq!(clmp, 0);
|
||||||
|
|
||||||
let vcc = match op {
|
let vcc = match op {
|
||||||
|
767 => {
|
||||||
|
let (s0, s1, s2): (u32, u32, u64) = (self.val(s0), self.val(s1), self.val(s2));
|
||||||
|
let (mul_result, overflow_mul) = (s0 as i64).overflowing_mul(s1 as i64);
|
||||||
|
let (ret, overflow_add) = mul_result.overflowing_add(s2 as i64);
|
||||||
|
let overflowed = overflow_mul || overflow_add;
|
||||||
|
if self.exec.read() {
|
||||||
|
self.vec_reg.write64(vdst, ret as u64);
|
||||||
|
}
|
||||||
|
overflowed
|
||||||
|
},
|
||||||
766 => {
|
766 => {
|
||||||
let (s0, s1, s2): (u32, u32, u64) = (self.val(s0), self.val(s1), self.val(s2));
|
let (s0, s1, s2): (u32, u32, u64) = (self.val(s0), self.val(s1), self.val(s2));
|
||||||
let (mul_result, overflow_mul) = (s0 as u64).overflowing_mul(s1 as u64);
|
let (mul_result, overflow_mul) = (s0 as u64).overflowing_mul(s1 as u64);
|
||||||
@@ -1246,7 +1257,7 @@ impl<'a> Thread<'a> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
let ret = match op {
|
let ret = match op {
|
||||||
257 | 259 | 299 | 260 | 261 | 264 | 272 | 392 | 426 | 531 | 537 | 540 | 551 | 567 | 796 => {
|
257 | 259 | 299 | 260 | 261 | 264 | 272 | 392 | 426 | 430 | 531 | 537 | 540 | 551 | 567 | 796 => {
|
||||||
let s0 = f32::from_bits(s0).negate(0, neg).absolute(0, abs);
|
let s0 = f32::from_bits(s0).negate(0, neg).absolute(0, abs);
|
||||||
let s1 = f32::from_bits(s1).negate(1, neg).absolute(1, abs);
|
let s1 = f32::from_bits(s1).negate(1, neg).absolute(1, abs);
|
||||||
let s2 = f32::from_bits(s2).negate(2, neg).absolute(2, abs);
|
let s2 = f32::from_bits(s2).negate(2, neg).absolute(2, abs);
|
||||||
@@ -1258,6 +1269,7 @@ impl<'a> Thread<'a> {
|
|||||||
272 => f32::max(s0, s1),
|
272 => f32::max(s0, s1),
|
||||||
299 => f32::mul_add(s0, s1, f32::from_bits(self.vec_reg[vdst])),
|
299 => f32::mul_add(s0, s1, f32::from_bits(self.vec_reg[vdst])),
|
||||||
426 => s0.recip(),
|
426 => s0.recip(),
|
||||||
|
430 => 1.0 / f32::sqrt(s0),
|
||||||
531 => f32::mul_add(s0, s1, s2),
|
531 => f32::mul_add(s0, s1, s2),
|
||||||
537 => f32::min(f32::min(s0, s1), s2),
|
537 => f32::min(f32::min(s0, s1), s2),
|
||||||
540 => f32::max(f32::max(s0, s1), s2),
|
540 => f32::max(f32::max(s0, s1), s2),
|
||||||
@@ -2625,6 +2637,14 @@ mod test_vop1 {
|
|||||||
assert_eq!(thread.vec_reg[3], 1071644672);
|
assert_eq!(thread.vec_reg[3], 1071644672);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_v_rsq_f32() {
|
||||||
|
let mut thread = _helper_test_thread();
|
||||||
|
thread.vec_reg[0] = f32::to_bits(4.0);
|
||||||
|
r(&vec![0x7E005D00, END_PRG], &mut thread);
|
||||||
|
assert_eq!(f32::from_bits(thread.vec_reg[0]), 0.5);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_v_frexp_exp_i32_f64() {
|
fn test_v_frexp_exp_i32_f64() {
|
||||||
[(3573412790272.0, 42), (69.0, 7), (2.0, 2), (f64::NEG_INFINITY, 0)]
|
[(3573412790272.0, 42), (69.0, 7), (2.0, 2), (f64::NEG_INFINITY, 0)]
|
||||||
|
|||||||
+1
-1
@@ -58,7 +58,7 @@ if __name__ == "__main__":
|
|||||||
GlobalCounters.kernel_count -= 1
|
GlobalCounters.kernel_count -= 1
|
||||||
|
|
||||||
if not getenv("NOOPT"): k.apply_opts(hand_coded_optimizations(k))
|
if not getenv("NOOPT"): k.apply_opts(hand_coded_optimizations(k))
|
||||||
p2 = get_program(k.get_optimized_ast(), k.opts)
|
p2 = get_program(k.ast, k.opts, k.applied_opts)
|
||||||
new_ei = replace(ei, prg=CompiledRunner(p2))
|
new_ei = replace(ei, prg=CompiledRunner(p2))
|
||||||
new_ei.run()
|
new_ei.run()
|
||||||
new_jit.append(new_ei)
|
new_jit.append(new_ei)
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user