Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e95e1d71c7 | ||
|
|
c8c5d33208 | ||
|
|
c05077fae3 | ||
|
|
bee0392c37 | ||
|
|
a6f6edd07d | ||
|
|
236c1378f9 | ||
|
|
88f816ed06 | ||
|
|
1c10560531 | ||
|
|
cf2d32d0a6 | ||
|
|
1265b2fe02 | ||
|
|
53d9316a56 | ||
|
|
648d516668 | ||
|
|
e961f7e344 | ||
|
|
fddd618915 | ||
|
|
0e71705a0a | ||
|
|
12138ced7c | ||
|
|
663b90035c | ||
|
|
aefc5314bc | ||
|
|
b1d9656470 | ||
|
|
22d7d03118 | ||
|
|
35fe2efe27 | ||
|
|
10ce1c0256 | ||
|
|
8978794730 | ||
|
|
5be1bc48e9 | ||
|
|
d70d86985e | ||
|
|
98cb7c2ce2 | ||
|
|
087bb34c68 | ||
|
|
9059d21042 | ||
|
|
c52382f547 | ||
|
|
8584df54e9 | ||
|
|
a5c19ea784 | ||
|
|
423b82ea6c | ||
|
|
39584d08ad | ||
|
|
619f984c36 | ||
|
|
7af4505519 | ||
|
|
a4fc4ffa6e | ||
|
|
6216501455 | ||
|
|
6517d1cf5c | ||
|
|
5f292390fd | ||
|
|
35ac30e688 | ||
|
|
10b16dbfab | ||
|
|
acab068c74 | ||
|
|
7b60d49432 | ||
|
|
1df0d2dc97 | ||
|
|
4b30ef6480 | ||
|
|
de1fdd8d3b | ||
|
|
5bb6b41b78 | ||
|
|
9d2df24d6b | ||
|
|
d120f97896 | ||
|
|
eeb411144f | ||
|
|
88c086bbd2 | ||
|
|
15c11fc848 | ||
|
|
d9bc8a978a | ||
|
|
d962ab5d89 | ||
|
|
7f64ad7a33 | ||
|
|
0cb6767465 | ||
|
|
ee17c7c9c8 | ||
|
|
76af84718a | ||
|
|
134eb61e1a | ||
|
|
4970927ec8 | ||
|
|
25bbd059df | ||
|
|
3a642601e8 | ||
|
|
f656882942 | ||
|
|
b9364f96b1 | ||
|
|
851866333c | ||
|
|
0cb58fbb4c | ||
|
|
35bbe178bd | ||
|
|
fc7f5919b5 | ||
|
|
2b03d34931 | ||
|
|
d6a0375974 | ||
|
|
2a2f303ae9 | ||
|
|
a6de1b8d75 | ||
|
|
48e808c20e | ||
|
|
043ae697c2 | ||
|
|
6d58fb1353 | ||
|
|
f90afa29b8 | ||
|
|
1a9f1c80a1 | ||
|
|
e865b046b1 | ||
|
|
1077159834 | ||
|
|
d28b145393 | ||
|
|
0cd5e64701 | ||
|
|
281a73ccf7 | ||
|
|
595ec65796 | ||
|
|
e6b34ef90d | ||
|
|
fafe5d63a7 | ||
|
|
d06d5e68b6 | ||
|
|
152a2eb30c | ||
|
|
4dc77b5a1a | ||
|
|
210cd657dd | ||
|
|
cf0d5dc470 | ||
|
|
f380027951 | ||
|
|
b4b73f92dd | ||
|
|
2950f66983 | ||
|
|
34bc149359 | ||
|
|
97c7b6b314 | ||
|
|
142bc0230e | ||
|
|
3eac6cfd4f | ||
|
|
d40425d257 | ||
|
|
2ec8d61e94 | ||
|
|
f9c9e39ab8 | ||
|
|
8d564b5e38 | ||
|
|
53aa5636cf | ||
|
|
42d5cfc3b0 | ||
|
|
9b86aea98b | ||
|
|
79196246cf | ||
|
|
ecb746e869 | ||
|
|
e40e27f59e | ||
|
|
981758fd04 | ||
|
|
b0ec7a0655 | ||
|
|
4fdf02965f | ||
|
|
994b77721c | ||
|
|
835ea9c2e3 | ||
|
|
cac8f0250c | ||
|
|
0782c6e8c4 | ||
|
|
9557b73cad | ||
|
|
b83b8005f9 | ||
|
|
9604d7bf89 | ||
|
|
2180aa19ad | ||
|
|
afd0e5d02f | ||
|
|
813e37916d | ||
|
|
68c9a110a2 | ||
|
|
0f019d7a94 | ||
|
|
2096306362 | ||
|
|
694f1d789d | ||
|
|
040c1f24e4 | ||
|
|
e7ea564df2 | ||
|
|
63addd091c | ||
|
|
a24c88ab08 | ||
|
|
710dbf5a12 | ||
|
|
a3cebb4469 | ||
|
|
5c0118fe9d | ||
|
|
013fd9886e | ||
|
|
4c4d0e6bc6 | ||
|
|
8b82ce0903 | ||
|
|
26933a97a3 | ||
|
|
89877fe064 | ||
|
|
b993a3ed39 | ||
|
|
cebc74f7bc | ||
|
|
2181ad1bc7 | ||
|
|
d2989f76e6 | ||
|
|
d5dff384eb | ||
|
|
4aac5568a6 | ||
|
|
6e86b59d21 | ||
|
|
f38b29ce6d | ||
|
|
f91b131ba2 | ||
|
|
735520b03e | ||
|
|
ccd49cfbc5 | ||
|
|
335819e1b7 | ||
|
|
312e394654 | ||
|
|
cdbf2f4a37 | ||
|
|
0ae7f479d3 | ||
|
|
e309b55b38 | ||
|
|
b19d61c251 | ||
|
|
e8de5282f0 | ||
|
|
a7ccd55372 | ||
|
|
f76b1125e2 | ||
|
|
acfb054103 | ||
|
|
0c285cd76f | ||
|
|
c0a517a553 | ||
|
|
b672fff564 | ||
|
|
e7a0d98277 | ||
|
|
d40b4c8f80 | ||
|
|
879d879985 | ||
|
|
761252be78 | ||
|
|
80df5039f8 | ||
|
|
b989756c07 | ||
|
|
e8024b0925 | ||
|
|
c13da418b5 | ||
|
|
f083995279 | ||
|
|
ea5e51c330 | ||
|
|
9f52e00bb4 | ||
|
|
f8a5defa98 | ||
|
|
9020cf91b5 | ||
|
|
dc2096c0e4 | ||
|
|
a258d3d31b | ||
|
|
f3369666e6 | ||
|
|
e8c087ed89 | ||
|
|
2c88e01736 | ||
|
|
1823c6997d | ||
|
|
d290b818d0 | ||
|
|
198d7715ee | ||
|
|
d2b94ca81b | ||
|
|
4755ded863 | ||
|
|
13bf772d96 | ||
|
|
17bce62e5f | ||
|
|
b620d86c54 | ||
|
|
791ba91dec | ||
|
|
d1279afff8 | ||
|
|
e684fdf60b | ||
|
|
1e2c9eaf89 | ||
|
|
cbd088bd13 | ||
|
|
f531ab957b | ||
|
|
58a467dd68 | ||
|
|
d0faf97893 | ||
|
|
d56b3e5e69 | ||
|
|
570b2c7aeb | ||
|
|
f07176da9b | ||
|
|
e0e67685d7 | ||
|
|
f3d139e90f | ||
|
|
cd15bfc3ce | ||
|
|
67d5f4dc39 | ||
|
|
890458fdbd | ||
|
|
3e8f2d99a9 | ||
|
|
fe2b6666e0 | ||
|
|
7989ca844c | ||
|
|
edb8d7a23c | ||
|
|
94e53444c6 | ||
|
|
5ab5084f7b | ||
|
|
47629536e2 | ||
|
|
68ca577919 | ||
|
|
29ebe92208 | ||
|
|
41b6cbb3ca | ||
|
|
0b22b64a10 | ||
|
|
e977d1cde5 | ||
|
|
545b38ec5f | ||
|
|
759557050a | ||
|
|
831842972f | ||
|
|
990fd22488 | ||
|
|
7024177f7d | ||
|
|
4d24032ea5 | ||
|
|
c1c6e3b6c9 | ||
|
|
bafdeca42f | ||
|
|
29c7d2f195 | ||
|
|
8035c10f37 | ||
|
|
bd168819f2 | ||
|
|
0203938af8 | ||
|
|
4fca994d0e | ||
|
|
452fa858f4 | ||
|
|
b0bf51f99f | ||
|
|
d0c9472cb3 | ||
|
|
7131685ae3 | ||
|
|
c71bd73acb | ||
|
|
ae2e14e3ed | ||
|
|
3c6f856f23 | ||
|
|
e02146943d | ||
|
|
a22a8142ac | ||
|
|
1ee2837d62 | ||
|
|
3ddf3f1fb4 | ||
|
|
9b31272cf0 | ||
|
|
2ab2f7d08d | ||
|
|
6e1d72d98a | ||
|
|
3c549e8ae3 | ||
|
|
06e6eadfaf | ||
|
|
e3001a0929 | ||
|
|
3431c62d41 | ||
|
|
8322f1b039 | ||
|
|
b3fe17ddeb | ||
|
|
c96c6a6b33 | ||
|
|
f293c9b5f4 | ||
|
|
8544b334e4 | ||
|
|
1b45ddcd17 | ||
|
|
3f1e4b953f | ||
|
|
3f09b32df3 | ||
|
|
d05ac813dc | ||
|
|
dcda5194df | ||
|
|
b78c3d4da8 | ||
|
|
7ac1580a31 | ||
|
|
e79ae18cae | ||
|
|
4c34d16a34 | ||
|
|
7857a73710 | ||
|
|
e052883de7 | ||
|
|
afc43dbba7 | ||
|
|
1f685c2882 | ||
|
|
8dd9b80d7a | ||
|
|
b2707c9b2e | ||
|
|
2dec93f588 | ||
|
|
17f58d2e11 | ||
|
|
b4eb3884cf | ||
|
|
21a1972921 | ||
|
|
b5c6d0e393 | ||
|
|
5b2351cbb9 | ||
|
|
764e7e12a7 | ||
|
|
fb8d085b5f | ||
|
|
2ae2bd2b46 | ||
|
|
7d0c2c7db8 | ||
|
|
d8cbf8d60c | ||
|
|
ddbf7de6dc | ||
|
|
471499cd78 | ||
|
|
f7622ebfca | ||
|
|
62822b6f73 | ||
|
|
fdb61cb854 | ||
|
|
466655bcda | ||
|
|
b780807e73 | ||
|
|
0d2eb95530 | ||
|
|
5ace7d455d | ||
|
|
b2ae57795f | ||
|
|
91a4ea9b38 | ||
|
|
c3b82f0170 | ||
|
|
09668df726 | ||
|
|
495ffbd028 | ||
|
|
b8ff9bc1d2 | ||
|
|
9754c5da55 | ||
|
|
4ed3027309 | ||
|
|
26cb5f6817 | ||
|
|
5f6be4dd53 | ||
|
|
91c9b29d47 | ||
|
|
b358714a9a | ||
|
|
38c56081ac | ||
|
|
b18accc64c | ||
|
|
fdcf9cd518 | ||
|
|
f1e11d8b38 | ||
|
|
1f2da71069 | ||
|
|
22bedf9b57 | ||
|
|
16f4cc9ff0 | ||
|
|
f6a86e8551 | ||
|
|
e68ba1c836 | ||
|
|
3c5530c29d | ||
|
|
dd5a05926c | ||
|
|
e570d2e1ca | ||
|
|
bf990a3cb3 | ||
|
|
38e89dd890 | ||
|
|
b31edf37bf | ||
|
|
003ca510fa | ||
|
|
42d9a02c08 | ||
|
|
f9bb796d29 | ||
|
|
c51651dba8 | ||
|
|
ebd9fc9530 | ||
|
|
868b172f05 | ||
|
|
2eca8a9ef2 | ||
|
|
1576ad9963 | ||
|
|
f33b5a8d99 | ||
|
|
724b787cd1 | ||
|
|
e73dcb8cbe | ||
|
|
2912239fe6 | ||
|
|
ddb59130f8 | ||
|
|
28242f02d1 | ||
|
|
80dc9795bd | ||
|
|
3cb149f4f4 | ||
|
|
6b41b5c589 | ||
|
|
04935ea718 | ||
|
|
e48422df38 | ||
|
|
7de51f78ac | ||
|
|
aca8c7e6f3 | ||
|
|
ee68d5ba8e | ||
|
|
d6646e151a | ||
|
|
6ddb03922a | ||
|
|
4dcb9d3e30 | ||
|
|
1aba411da9 | ||
|
|
a707d4bea1 | ||
|
|
2ccc7456ca | ||
|
|
09167efdb5 | ||
|
|
3476d2f279 | ||
|
|
2ca5356429 | ||
|
|
31a658e558 | ||
|
|
31017120fd | ||
|
|
b7de42f70d | ||
|
|
18d055a390 | ||
|
|
c869dd8b8f | ||
|
|
6dfe9951e1 | ||
|
|
1d1aba812b | ||
|
|
31b71483c4 | ||
|
|
b74a3c5106 | ||
|
|
fb42872259 | ||
|
|
ab09faa15e | ||
|
|
54507f417e | ||
|
|
dab3b965cb | ||
|
|
4e0d0ab8d2 | ||
|
|
0b0180406b | ||
|
|
12b39a74b4 | ||
|
|
61177cd1c8 | ||
|
|
1a9719c1c8 | ||
|
|
ac6692d3e4 | ||
|
|
3a93aaf9e2 | ||
|
|
da185342ce | ||
|
|
9bb2e00cb6 | ||
|
|
bec43c9f8a | ||
|
|
593bf50759 | ||
|
|
6772e0c197 | ||
|
|
6a0b171be4 | ||
|
|
d394b80ac8 | ||
|
|
2a4cd479e2 | ||
|
|
e86e6b2faa | ||
|
|
45d671a4a8 | ||
|
|
ced662fc27 | ||
|
|
f6dabc2fe9 | ||
|
|
6a9bbbc886 | ||
|
|
60b8246bc3 | ||
|
|
ce9d87597b | ||
|
|
5917e6b138 | ||
|
|
38b63f95af | ||
|
|
d735055e6f | ||
|
|
e880e29f2b | ||
|
|
4c2026bf9a | ||
|
|
3be81cb54e | ||
|
|
792962ecc9 | ||
|
|
732eaee4d7 | ||
|
|
36274bed49 | ||
|
|
711892a0a2 | ||
|
|
01b8991c5a | ||
|
|
22a7264e9a | ||
|
|
73a911890b | ||
|
|
1a73fa0b03 | ||
|
|
c32e3f3ea5 | ||
|
|
e461ec0037 | ||
|
|
49d000c0c9 | ||
|
|
c89e482f85 | ||
|
|
384e124490 | ||
|
|
774d9be357 | ||
|
|
3ad6169f18 | ||
|
|
2232eb35d1 | ||
|
|
c0bedd2587 | ||
|
|
da61398835 | ||
|
|
f6a7a5278a | ||
|
|
3c2fd560aa | ||
|
|
1d5f06223a | ||
|
|
2b3f443f6b | ||
|
|
1383f64a5f | ||
|
|
fb27a771f8 | ||
|
|
514d182b7f | ||
|
|
322e7157e0 | ||
|
|
ed0b890e75 | ||
|
|
b4d4e489bf | ||
|
|
b29613b758 | ||
|
|
9255e54acb | ||
|
|
5e013f6e2f | ||
|
|
5691ffb160 | ||
|
|
bc01b9ac1c | ||
|
|
620941eb28 | ||
|
|
0cdfb9af41 | ||
|
|
3854afdc3e | ||
|
|
af621f8590 | ||
|
|
21057d0064 | ||
|
|
3223e71b30 | ||
|
|
e17c72148e | ||
|
|
38e9fabdd5 | ||
|
|
4896815067 | ||
|
|
3d18099262 | ||
|
|
2bc01a00e8 | ||
|
|
21ac476b17 | ||
|
|
0d5f7671e0 | ||
|
|
be89eb07c8 | ||
|
|
9f140b7698 | ||
|
|
ff1f8ef400 | ||
|
|
10311c48b7 | ||
|
|
09482fb64f | ||
|
|
1a2726fa6b | ||
|
|
25bb33ad45 | ||
|
|
b183782e83 | ||
|
|
f6f11d31d3 | ||
|
|
61030e4f4b | ||
|
|
1ec65c3ffe | ||
|
|
6bc7132c99 | ||
|
|
2d8913c5c6 | ||
|
|
d80215ed9d | ||
|
|
f996c2892f | ||
|
|
6f2b296b64 | ||
|
|
ee6f58003b | ||
|
|
e171c4000e | ||
|
|
c5b81d04c7 | ||
|
|
3c85332661 | ||
|
|
fd53d434c7 | ||
|
|
9c1ecb8972 | ||
|
|
e125005cec | ||
|
|
aa21debc08 | ||
|
|
08d966db01 | ||
|
|
7bb245d179 | ||
|
|
2ca877d78a | ||
|
|
15e268d6df | ||
|
|
0ebfb78570 | ||
|
|
bb7356bcaa | ||
|
|
8827bd3a3b | ||
|
|
8ff19dda22 | ||
|
|
29faea1862 | ||
|
|
969e929a48 | ||
|
|
bcb45d906d | ||
|
|
165b9fb3f3 | ||
|
|
e586ed4767 | ||
|
|
6a39573267 | ||
|
|
010f63d219 | ||
|
|
f6934e5f14 | ||
|
|
0c1900a988 | ||
|
|
64de57b09e | ||
|
|
d1c0f1270d | ||
|
|
25c12258e4 | ||
|
|
0505d48287 | ||
|
|
1d11f61c36 | ||
|
|
2ac236ebfe | ||
|
|
4aaa7d28ea | ||
|
|
4c5e82c065 | ||
|
|
05676de2d9 | ||
|
|
67dc9bc135 | ||
|
|
705e576417 | ||
|
|
17891653cd | ||
|
|
29cbc9e723 | ||
|
|
6dae5698ef | ||
|
|
5458d05cd8 | ||
|
|
f862d9f691 | ||
|
|
04c9eb49d0 | ||
|
|
2a04be0386 | ||
|
|
e15c0419c0 | ||
|
|
d69455a466 | ||
|
|
bd3dc788cc | ||
|
|
064101d04f | ||
|
|
74a2a5822a | ||
|
|
73f78a10a2 | ||
|
|
45e0f4b369 | ||
|
|
b2d8f26e46 | ||
|
|
94afe8236c | ||
|
|
479a35d94e | ||
|
|
1ae16bdb2c | ||
|
|
bbaa777767 | ||
|
|
ad80a7d638 | ||
|
|
7beed7cae6 | ||
|
|
563e2ba2c6 | ||
|
|
f5e0df390c | ||
|
|
27a3be0287 | ||
|
|
b9418450ac | ||
|
|
d856989120 | ||
|
|
f86dd55145 | ||
|
|
b2e9607362 | ||
|
|
be244560b2 | ||
|
|
96b058c5fa | ||
|
|
be83e7515b | ||
|
|
a5f159b2c7 | ||
|
|
5d89fed2a6 | ||
|
|
5dd2afeab1 | ||
|
|
20d15c8023 | ||
|
|
932770771b | ||
|
|
4d36e76cbc | ||
|
|
9854084136 | ||
|
|
ceec51d96c | ||
|
|
6b667b1237 | ||
|
|
2b5293ddfc | ||
|
|
1015a00506 | ||
|
|
c56ee8bdee | ||
|
|
5778a4131c | ||
|
|
89d5772f55 | ||
|
|
da2f11a9c4 | ||
|
|
e05586c4b2 | ||
|
|
446a1e23d7 | ||
|
|
4ac9925dad | ||
|
|
c00a8a10dd | ||
|
|
6e7dc9c236 | ||
|
|
2b5458e852 | ||
|
|
5c5a241e01 | ||
|
|
f7e9700aae | ||
|
|
9eb1907151 | ||
|
|
897def2cac | ||
|
|
b933b23d5c | ||
|
|
56dddf9708 | ||
|
|
b5e9fd0b2c | ||
|
|
b1040523b2 | ||
|
|
ea8878bc14 | ||
|
|
c58aab0b00 | ||
|
|
c4b0693a4d | ||
|
|
ffd6e693de | ||
|
|
b9b5a93f0f | ||
|
|
dfbb50cd6a | ||
|
|
054a35312d | ||
|
|
9571de8757 | ||
|
|
3562aa5aae | ||
|
|
919a26fe41 | ||
|
|
d4a31f02e0 | ||
|
|
e38b18e9eb | ||
|
|
0ad3e8b8e9 | ||
|
|
f44dfb3e7a | ||
|
|
a33beb6ebf | ||
|
|
93e8ad1aa7 | ||
|
|
43ac63f2e7 | ||
|
|
6029fad989 | ||
|
|
11db9d29ad | ||
|
|
d3d7e7bf1f | ||
|
|
edd4a87fb0 | ||
|
|
27bba1a03a | ||
|
|
4ae31cd1d5 | ||
|
|
06ca6428b6 | ||
|
|
62e9963cf7 | ||
|
|
2d2f94ddb6 | ||
|
|
a4262996e5 | ||
|
|
149e466dbe | ||
|
|
c48541dedc | ||
|
|
9f939447f2 | ||
|
|
539d129178 | ||
|
|
21d0f32047 | ||
|
|
b35512ce82 | ||
|
|
b13b7d8033 | ||
|
|
7459c09218 | ||
|
|
64d1bac7d6 | ||
|
|
f7f1dc3f4e | ||
|
|
853c4c1e7b | ||
|
|
ed26c177b2 | ||
|
|
6443d1512d | ||
|
|
4c6c3d04ce | ||
|
|
af44583050 | ||
|
|
8fa802e35b | ||
|
|
fc0ad03008 | ||
|
|
5130841bef | ||
|
|
bfbb4a6279 | ||
|
|
1cf430f7bc | ||
|
|
57074b3268 | ||
|
|
5035ce5474 | ||
|
|
734b28ed2d | ||
|
|
4cbcb7887e | ||
|
|
6d32595e2b | ||
|
|
b5cab7e0f5 | ||
|
|
472f394788 | ||
|
|
5e97e66146 | ||
|
|
589815f6ab | ||
|
|
76a1c67d87 | ||
|
|
784a053793 | ||
|
|
9a6838d349 | ||
|
|
deffbaba7f | ||
|
|
7deec2c14e | ||
|
|
cc12ff36a9 | ||
|
|
b35c472bb1 | ||
|
|
946aef6216 | ||
|
|
a804755e6e | ||
|
|
50881c0b31 | ||
|
|
588ad83771 | ||
|
|
9f5a7e64b6 | ||
|
|
0083435764 | ||
|
|
398726e830 | ||
|
|
f80127db0e | ||
|
|
c649f63e7e | ||
|
|
4d98d8ad31 | ||
|
|
675dbedb82 | ||
|
|
c5d4b87375 | ||
|
|
3513cb4df9 | ||
|
|
432a0bcd06 | ||
|
|
eeb48ceb96 | ||
|
|
f8d9f8f773 | ||
|
|
dfb6d3626e | ||
|
|
ca894f081b | ||
|
|
d960774ae6 | ||
|
|
707bcb2827 | ||
|
|
9e654c4ec8 | ||
|
|
a2b20b46bc | ||
|
|
9aad69d856 | ||
|
|
06242c200a | ||
|
|
ea59a99426 | ||
|
|
de2ccc03a8 | ||
|
|
53b7644c15 | ||
|
|
dac59bb8d3 | ||
|
|
bc67689068 | ||
|
|
bde549cb36 | ||
|
|
deb1581e26 | ||
|
|
f72e354ee6 | ||
|
|
6fdfa12e50 | ||
|
|
34a7266bc2 | ||
|
|
92fb0c267e | ||
|
|
4ac82584dc | ||
|
|
88b750a018 | ||
|
|
7a1df80f4e | ||
|
|
3002bd3df5 | ||
|
|
91ee0711f0 | ||
|
|
756c70a4a0 | ||
|
|
083dd6a3ef | ||
|
|
ec7fc97857 | ||
|
|
f7db44e750 | ||
|
|
8dc8a8bfd3 |
@@ -10,29 +10,72 @@ references:
|
||||
run:
|
||||
name: Install Dependences
|
||||
command: |
|
||||
pip install "$TORCH_VERSION" --user
|
||||
# this is temporal fix til test-tube is not merged and released
|
||||
pip install -r requirements.txt --user
|
||||
sudo pip install pytest pytest-cov pytest-flake8
|
||||
pip install -r ./tests/requirements.txt --user
|
||||
sudo apt-get update && sudo apt-get install -y cmake
|
||||
pip install "$TORCH_VERSION"
|
||||
pip install -r requirements.txt -q
|
||||
sudo pip install pytest -q
|
||||
pip install -r ./tests/requirements-devel.txt -q
|
||||
|
||||
tests_format: &tests_format
|
||||
tests: &tests
|
||||
run:
|
||||
name: Tests and formating
|
||||
name: Testing
|
||||
command: |
|
||||
python --version ; pip --version ; pip list
|
||||
py.test pytorch_lightning tests pl_examples -v --doctest-modules --junitxml=test-reports/pytest_junit.xml --flake8
|
||||
no_output_timeout: 15m
|
||||
py.test pytorch_lightning tests -v --doctest-modules --junitxml=test-reports/pytest_junit.xml
|
||||
no_output_timeout: 30m
|
||||
|
||||
examples: &examples
|
||||
run:
|
||||
name: PL Examples
|
||||
command: |
|
||||
pip install -r ./pl_examples/requirements.txt --user
|
||||
python --version ; pip --version ; pip list
|
||||
py.test pl_examples -v --doctest-modules --junitxml=test-reports/pytest_junit.xml
|
||||
no_output_timeout: 20m
|
||||
|
||||
install_pkg: &install_pkg
|
||||
run:
|
||||
name: Install package
|
||||
command: |
|
||||
virtualenv vEnv ; source vEnv/bin/activate
|
||||
pip install --editable . ; cd .. & python -c "import pytorch_lightning ; print(pytorch_lightning.__version__)"
|
||||
deactivate ; rm -rf vEnv
|
||||
|
||||
create_pkg: &create_pkg
|
||||
run:
|
||||
name: Create package
|
||||
command: |
|
||||
sudo pip install twine==1.13.0
|
||||
python setup.py sdist
|
||||
twine check dist/*
|
||||
python setup.py clean
|
||||
|
||||
format: &format
|
||||
run:
|
||||
name: Formatting
|
||||
command: |
|
||||
python --version ; pip --version
|
||||
sudo pip install flake8 -q
|
||||
pip list
|
||||
flake8 .
|
||||
|
||||
make_docs: &make_docs
|
||||
run:
|
||||
name: Make Documentation
|
||||
command: |
|
||||
# sudo apt-get install pandoc
|
||||
pip install -r requirements.txt --user
|
||||
# First run the same pipeline as Read-The-Docs
|
||||
sudo apt-get update && sudo apt-get install -y cmake
|
||||
sudo pip install -r docs/requirements.txt
|
||||
# sphinx-apidoc -o ./docs/source ./pytorch_lightning **/test_* --force --follow-links
|
||||
cd docs; make clean ; make html
|
||||
cd docs; make clean; make html --debug --jobs 2 SPHINXOPTS="-W"
|
||||
|
||||
test_docs: &test_docs
|
||||
run:
|
||||
name: Testing Documentation
|
||||
command: |
|
||||
# Second run examples in docs
|
||||
sudo apt-get update && sudo apt-get install -y cmake
|
||||
sudo pip install -r docs/requirements.txt
|
||||
cd docs; make doctest; make coverage
|
||||
|
||||
jobs:
|
||||
|
||||
@@ -42,49 +85,114 @@ jobs:
|
||||
steps:
|
||||
- checkout
|
||||
- *make_docs
|
||||
- store_artifacts:
|
||||
# allows us to preview the generated html pages
|
||||
path: docs/build/html/
|
||||
destination: html
|
||||
|
||||
Formatting:
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
- TORCH_VERSION: "torch"
|
||||
steps:
|
||||
- checkout
|
||||
- *format
|
||||
|
||||
PyTorch:
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
- TORCH_VERSION: "torch"
|
||||
steps: &steps
|
||||
- checkout
|
||||
|
||||
#- restore_cache:
|
||||
# keys:
|
||||
# # when lock file changes, use increasingly general patterns to restore cache
|
||||
# - pip-packages--{{ .Environment.CIRCLE_JOB }}
|
||||
# - pip-packages--
|
||||
- *install_deps
|
||||
- *tests_format
|
||||
|
||||
#- save_cache:
|
||||
# key: pip-packages--{{ .Environment.CIRCLE_JOB }}
|
||||
# paths:
|
||||
# # this path depends on where pipenv creates a virtualenv
|
||||
# - "~/.cache/pip"
|
||||
# - "/usr/local/lib/python3.6/site-packages"
|
||||
# - "/usr/local/lib/site-python"
|
||||
- *tests
|
||||
- store_test_results:
|
||||
path: test-reports
|
||||
- store_artifacts:
|
||||
path: test-reports
|
||||
|
||||
PyTorch-v1.1:
|
||||
PyTorch-v1_1:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
- TORCH_VERSION: "torch>=1.1, <1.2"
|
||||
steps: *steps
|
||||
|
||||
PyTorch-v1.2:
|
||||
PyTorch-v1_2:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
- TORCH_VERSION: "torch>=1.2, <1.3"
|
||||
steps: *steps
|
||||
|
||||
PyTorch-v1.3:
|
||||
PyTorch-v1_3:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
- TORCH_VERSION: "torch>=1.3, <1.4"
|
||||
steps: *steps
|
||||
|
||||
PyTorch-v1_4:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
- TORCH_VERSION: "torch>=1.4, <1.5"
|
||||
steps: *steps
|
||||
|
||||
PyTorch-v1_5:
|
||||
docker:
|
||||
- image: circleci/python:3.6
|
||||
environment:
|
||||
- TORCH_VERSION: "torch>=1.5, <1.6"
|
||||
steps: *steps
|
||||
|
||||
Examples:
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
environment:
|
||||
- TORCH_VERSION: "torch"
|
||||
- SPHINX_MOCK_REQUIREMENTS: 0
|
||||
steps:
|
||||
- checkout
|
||||
- *install_deps
|
||||
- *test_docs
|
||||
- *examples
|
||||
|
||||
Install-pkg:
|
||||
docker:
|
||||
- image: circleci/python:3.7
|
||||
steps:
|
||||
- checkout
|
||||
- *create_pkg
|
||||
- *install_pkg
|
||||
|
||||
#orbs:
|
||||
# python: circleci/python@0.2.1
|
||||
|
||||
workflows:
|
||||
version: 2
|
||||
build:
|
||||
jobs:
|
||||
- Formatting
|
||||
- Build-Docs
|
||||
- PyTorch-v1.1
|
||||
- PyTorch-v1.2
|
||||
- PyTorch-v1.3
|
||||
- PyTorch-v1_1
|
||||
- PyTorch-v1_2
|
||||
- PyTorch-v1_3
|
||||
- PyTorch-v1_4
|
||||
- PyTorch-v1_5
|
||||
- Install-pkg
|
||||
- Examples
|
||||
|
||||
@@ -2,9 +2,15 @@
|
||||
# Validation check:
|
||||
# $ curl --data-binary @.codecov.yml https://codecov.io/validate
|
||||
|
||||
|
||||
# https://docs.codecov.io/docs/codecovyml-reference
|
||||
codecov:
|
||||
bot: "codecov-io"
|
||||
strict_yaml_branch: "yaml-config"
|
||||
require_ci_to_pass: yes
|
||||
notify:
|
||||
require_ci_to_pass: yes
|
||||
# after_n_builds: 2
|
||||
wait_for_ci: yes
|
||||
|
||||
coverage:
|
||||
precision: 0 # 2 = xx.xx%, 0 = xx%
|
||||
@@ -16,7 +22,7 @@ coverage:
|
||||
default:
|
||||
against: auto
|
||||
target: 99% # specify the target coverage for each commit status
|
||||
threshold: 20% # allow this little decrease on project
|
||||
threshold: 30% # allow this little decrease on project
|
||||
# https://github.com/codecov/support/wiki/Filtering-Branches
|
||||
# branches: master
|
||||
if_ci_failed: error
|
||||
@@ -24,7 +30,7 @@ coverage:
|
||||
patch:
|
||||
default:
|
||||
against: auto
|
||||
target: 40% # specify the target "X%" coverage to hit
|
||||
target: 50% # specify the target "X%" coverage to hit
|
||||
# threshold: 50% # allow this much decrease on patch
|
||||
changes: false
|
||||
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
# https://docs.drone.io/pipeline/docker/examples/languages/python/#python-example
|
||||
|
||||
kind: pipeline
|
||||
type: docker
|
||||
name: torch-GPU
|
||||
|
||||
steps:
|
||||
- name: testing
|
||||
image: pytorchlightning/pytorch_lightning:devel-pt_1_4
|
||||
|
||||
environment:
|
||||
SLURM_LOCALID: 0
|
||||
CODECOV_TOKEN:
|
||||
from_secret: codecov_token
|
||||
HOROVOD_GPU_ALLREDUCE: NCCL
|
||||
HOROVOD_GPU_BROADCAST: NCCL
|
||||
HOROVOD_WITH_PYTORCH: 1
|
||||
HOROVOD_WITHOUT_TENSORFLOW: 1
|
||||
HOROVOD_WITHOUT_MXNET: 1
|
||||
HOROVOD_WITH_GLOO: 1
|
||||
HOROVOD_WITHOUT_MPI: 1
|
||||
|
||||
#volumes:
|
||||
# # Mount pip cache from host
|
||||
# - name: pip_cache
|
||||
# path: /opt/conda/lib/python3.7/site-packages
|
||||
|
||||
commands:
|
||||
- export PATH="$PATH:/root/.local/bin"
|
||||
- python --version
|
||||
- pip install pip -U
|
||||
- pip --version
|
||||
- nvidia-smi
|
||||
#- bash ./tests/install_AMP.sh
|
||||
- apt-get update && apt-get install -y cmake
|
||||
- pip install -r requirements.txt --user -q
|
||||
- pip install -r ./tests/requirements-devel.txt --user -q
|
||||
#- pip install -r ./docs/requirements.txt --user -q
|
||||
- pip list
|
||||
- python -c "import torch ; print(' & '.join([torch.cuda.get_device_name(i) for i in range(torch.cuda.device_count())]) if torch.cuda.is_available() else 'only CPU')"
|
||||
- coverage run --source pytorch_lightning -m py.test pytorch_lightning tests benchmarks -v --doctest-modules # --flake8
|
||||
#- cd docs; make doctest; make coverage
|
||||
- coverage report
|
||||
- codecov --token $CODECOV_TOKEN # --pr $DRONE_PULL_REQUEST --build $DRONE_BUILD_NUMBER --branch $DRONE_BRANCH --commit $DRONE_COMMIT --tag $DRONE_TAG
|
||||
- python tests/collect_env_details.py
|
||||
|
||||
trigger:
|
||||
branch:
|
||||
- master
|
||||
event:
|
||||
include:
|
||||
- push
|
||||
- pull_request
|
||||
|
||||
#volumes:
|
||||
# - name: pip_cache
|
||||
# host:
|
||||
# path: /tmp/cache/drone/pip
|
||||
@@ -6,7 +6,7 @@ We're currently recruiting for a team of 5 core maintainers.
|
||||
As a core maintainer you will have a strong say in the direction of the project. Big changes will require a majority of maintainers to agree.
|
||||
|
||||
### Code of conduct
|
||||
First and foremost, you'll be evaluated against [these core values](https://github.com/williamFalcon/pytorch-lightning/blob/master/.github/CONTRIBUTING.md). Any code we commit or feature we add needs to align with those core values.
|
||||
First and foremost, you'll be evaluated against [these core values](https://github.com/PyTorchLightning/pytorch-lightning/blob/master/.github/CONTRIBUTING.md). Any code we commit or feature we add needs to align with those core values.
|
||||
|
||||
### The bar for joining the team
|
||||
Lightning is being used to solve really hard problems at the top AI labs in the world. As such, the bar for adding team members is extremely high. Candidates must have solid engineering skills, have a good eye for user experience, and must be a power user of Lightning and PyTorch.
|
||||
|
||||
@@ -1,53 +1,186 @@
|
||||
# Contributing
|
||||
Welcome to the PyTorch Lightning community! We're building the most advanced research platform on the planet to implement the latest, best practices that the amazing PyTorch team rolls out!
|
||||
|
||||
## Main Core Value: One less thing to remember
|
||||
Simplify the API as much as possible from the user perspective. Any additions or improvements should minimize things the user needs to remember.
|
||||
## Main Core Value: One less thing to remember
|
||||
|
||||
For example: One benefit of the validation_step is that the user doesn't have to remember to set the model to .eval(). This avoids all sorts of subtle errors the user could make.
|
||||
Simplify the API as much as possible from the user perspective.
|
||||
Any additions or improvements should minimize things the user needs to remember.
|
||||
|
||||
## Lightning Design Principles
|
||||
We encourage all sorts of contributions you're interested in adding! When coding for lightning, please follow these principles.
|
||||
#### No PyTorch Interference
|
||||
We don't want to add any abstractions on top of pure PyTorch. This gives researchers all the control they need without having to learn yet another framework.
|
||||
For example: One benefit of the validation_step is that the user doesn't have to remember to set the model to .eval().
|
||||
This avoids all sorts of subtle errors the user could make.
|
||||
|
||||
#### Simple Internal Code
|
||||
It's useful for users to look at the code and understand very quickly what's happening. Many users won't be engineers. Thus we need to value clear, simple code over condensed ninja moves. While that's super cool, this isn't the project for that :)
|
||||
## Lightning Design Principles
|
||||
We encourage all sorts of contributions you're interested in adding! When coding for lightning, please follow these principles.
|
||||
|
||||
#### No PyTorch Interference
|
||||
We don't want to add any abstractions on top of pure PyTorch.
|
||||
This gives researchers all the control they need without having to learn yet another framework.
|
||||
|
||||
#### Force User Decisions To Best Practices
|
||||
There are 1,000 ways to do something. However, something eventually becomes standard practice that everyone does. Thus we pick one way of doing it and force everyone to do it this way. A good example is accumulated gradients. There are many ways to implement, we just pick one and force users to use that one. A bad forced decision would be to make users use a specific library to do something.
|
||||
#### Simple Internal Code
|
||||
It's useful for users to look at the code and understand very quickly what's happening.
|
||||
Many users won't be engineers. Thus we need to value clear, simple code over condensed ninja moves.
|
||||
While that's super cool, this isn't the project for that :)
|
||||
|
||||
#### Force User Decisions To Best Practices
|
||||
There are 1,000 ways to do something. However, something eventually becomes standard practice that everyone does.
|
||||
Thus we pick one way of doing it and force everyone to do it this way.
|
||||
A good example is accumulated gradients.
|
||||
There are many ways to implement, we just pick one and force users to use that one.
|
||||
A bad forced decision would be to make users use a specific library to do something.
|
||||
|
||||
When something becomes a best practice, we add it to the framework. This likely looks like code in utils or in the model file that everyone keeps adding over and over again across projects. When this happens, bring that code inside the trainer and add a flag for it.
|
||||
|
||||
#### Simple External API
|
||||
What makes sense to you may not make sense to others. Create an issue with an API change suggestion and validate that it makes sense for others. Treat code changes how you treat a startup: validate that it's a needed feature, then add if it makes sense for many people.
|
||||
#### Simple External API
|
||||
What makes sense to you may not make sense to others. Create an issue with an API change suggestion and validate that it makes sense for others.
|
||||
Treat code changes how you treat a startup: validate that it's a needed feature, then add if it makes sense for many people.
|
||||
|
||||
#### Backward-compatible API
|
||||
#### Backward-compatible API
|
||||
We all hate updating our deep learning packages because we don't want to refactor a bunch of stuff. In Lightning, we make sure every change we make which could break an API is backwards compatible with good deprecation warnings.
|
||||
|
||||
You shouldn't be afraid to upgrade Lightning :)
|
||||
**You shouldn't be afraid to upgrade Lightning :)**
|
||||
|
||||
#### Gain User Trust
|
||||
As a researcher you can't have any part of your code going wrong. So, make thorough tests that ensure an implementation of a new trick or subbtle change is correct.
|
||||
#### Gain User Trust
|
||||
As a researcher you can't have any part of your code going wrong. So, make thorough tests that ensure an implementation of a new trick or subbtle change is correct.
|
||||
|
||||
#### Interoperability
|
||||
#### Interoperability
|
||||
Have a favorite feature from other libraries like fast.ai or transformers? Those should just work with lightning as well. Grab your favorite model or learning rate scheduler from your favorite library and run it in Lightning.
|
||||
|
||||
## Contribution Types
|
||||
Currently looking for help implementing new features or adding bug fixes.
|
||||
---
|
||||
|
||||
A lot of good work has already been done in project mechanics (requirements.txt, setup.py, pep8, badges, ci, etc...) we're in a good state there thanks to all the early contributors (even pre-beta release)!
|
||||
## Contribution Types
|
||||
Currently looking for help implementing new features or adding bug fixes.
|
||||
|
||||
## Bug Fixes:
|
||||
1. Submit a github issue.
|
||||
2. Fix it.
|
||||
3. Submit a PR!
|
||||
A lot of good work has already been done in project mechanics (requirements.txt, setup.py, pep8, badges, ci, etc...) we're in a good state there thanks to all the early contributors (even pre-beta release)!
|
||||
|
||||
## New Features:
|
||||
1. Submit a github issue.
|
||||
2. We'll agree on the feature scope.
|
||||
3. Submit a PR! (with updated docs and tests 🙃).
|
||||
### Bug Fixes:
|
||||
1. Submit a github issue - try to decried what happen so other can reproduce it too.
|
||||
2. Try to ix it or recommend a solution...
|
||||
3. Submit a PR!
|
||||
|
||||
## Coding Styleguide
|
||||
1. Test the code with flake8.
|
||||
2. Use f-strings.
|
||||
|
||||
### New Features:
|
||||
1. Submit a github issue - describe what is motivation of such feature (plus an use-case).
|
||||
2. Let's discuss to agree on the feature scope.
|
||||
3. Submit a PR! (with updated docs and tests 🙃).
|
||||
|
||||
---
|
||||
|
||||
## Guidelines
|
||||
|
||||
### Coding Style
|
||||
|
||||
1. Use f-strings for output formation (except logging when we stay with lazy `logging.info("Hello %s!`, name).
|
||||
2. Test the code with flake8, run locally PEP8 fixes:
|
||||
```
|
||||
autopep8 -v -r --max-line-length 120 --in-place .
|
||||
```
|
||||
|
||||
### Documentation
|
||||
|
||||
We are using Sphinx with Napoleon extension.
|
||||
Moreover we set Google style to follow with type convention.
|
||||
|
||||
- [Napoleon formatting with Google style](https://sphinxcontrib-napoleon.readthedocs.io/en/latest/example_google.html)
|
||||
- [ReStructured Text (reST)](https://docs.pylonsproject.org/projects/docs-style-guide/)
|
||||
- [Paragraph-level markup](https://www.sphinx-doc.org/en/1.5/markup/para.html)
|
||||
|
||||
See following short example of a sample function taking one position string and optional
|
||||
|
||||
```python
|
||||
from typing import Optional
|
||||
|
||||
def my_func(param_a: int, param_b: Optional[float] = None) -> str:
|
||||
"""Sample function.
|
||||
|
||||
Args:
|
||||
param_a: first parameter
|
||||
param_b: second parameter
|
||||
|
||||
Return:
|
||||
sum of both numbers
|
||||
|
||||
Example:
|
||||
Sample doctest example...
|
||||
>>> my_func(1, 2)
|
||||
3
|
||||
|
||||
.. note:: If you want to add something.
|
||||
"""
|
||||
p = param_b if param_b else 0
|
||||
return str(param_a + p)
|
||||
```
|
||||
|
||||
When updating the docs make sure to build them first locally and visually inspect the html files (in the browser) for
|
||||
formatting errors. In certain cases, a missing blank line or a wrong indent can lead to a broken layout.
|
||||
Run these commands
|
||||
```bash
|
||||
cd docs
|
||||
pip install -r requirements.txt
|
||||
make html
|
||||
```
|
||||
and open `docs/build/html/index.html` in your browser.
|
||||
|
||||
When you send a PR the continuous integration will run tests and build the docs. You can access a preview of the html pages in the
|
||||
_Artifacts_ tab in CircleCI when you click on the task named _ci/circleci: Build-Docs_ at the bottom of the PR page.
|
||||
|
||||
### Testing
|
||||
|
||||
Test your work locally to speed up your work since so you can focus only in particular (failing) test-cases.
|
||||
To setup a local development environment, install both local and test dependencies:
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
pip install -r tests/requirements-devel.txt
|
||||
```
|
||||
|
||||
You can run the full test-case in your terminal via this bash script:
|
||||
|
||||
```bash
|
||||
bash .run_local_tests.sh
|
||||
```
|
||||
|
||||
Note: if your computer does not have multi-GPU nor TPU these tests are skipped.
|
||||
|
||||
For convenience, you can use also your own CircleCI building which will be triggered with each commit.
|
||||
This is useful if you do not test against all required dependencies version.
|
||||
To do so, login to [CircleCI](https://app.circleci.com/) and enable your forked project in the dashboard. It will just work after that.
|
||||
|
||||
### Pull Request
|
||||
|
||||
We welcome any useful contribution! For convinece here's a recommended workflow:
|
||||
|
||||
0. Think about what you want to do - fix a bug, repair docs, etc.
|
||||
1. Start your work locally (usually until you need our CI testing)
|
||||
- create a branch and prepare your changes
|
||||
- hint: do not work with your master directly, it may become complicated when you need to rebase
|
||||
- hint: give your PR a good name! it will be useful later when you may work on multiple tasks/PRs
|
||||
2. Create a "Draft PR" which is clearly marked which lets us know you don't need feedback yet.
|
||||
3. When you feel like you are ready for integrating your work, turn your PR to "Ready for review".
|
||||
4. Use tags in PR name for following cases:
|
||||
- **[blocked by #<number>]** if you work is depending on others changes
|
||||
- **[wip]** when you start to re-edit your work, mark it so no one will accidentally merge it in meantime
|
||||
|
||||
### Question & Answer
|
||||
|
||||
1. **How can I help/contribute?**
|
||||
|
||||
All help is very welcome - reporting bug, solving issues and preparing bug fixes. To solve some issues you can start with label [good first issue](https://github.com/PyTorchLightning/pytorch-lightning/issues?q=is%3Aopen+is%3Aissue+label%3A%22good+first+issue%22) or chose something close to your domain with label [help wanted](https://github.com/PyTorchLightning/pytorch-lightning/issues?q=is%3Aopen+is%3Aissue+label%3A%22help+wanted%22). Before you start to implement anything check that the issue description that it is clear and self-assign the task to you (if it is not possible, just comment that you take it and we assign it to you...).
|
||||
|
||||
2. **Is there a recommendation for branch names?**
|
||||
|
||||
We do not rely on the name convention so far you are working with your own fork. Anyway it would be nice to follow this convention `<type>/<issue-id>_<short-name>` where the types are: `bugfix`, `feaure`, `docs`, `tests`, ...
|
||||
|
||||
3. **How to rebase my PR?**
|
||||
|
||||
We recommend to create a PR in separate branch different from `master`, especially if you plan to submit several changes and do not want to wait until the fist one is resolved (we can work on them in parallel). Update your master with upstream (assuming you have already set [upstream](https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/configuring-a-remote-for-a-fork))
|
||||
```bash
|
||||
git fetch --all --prune
|
||||
git checkout master
|
||||
git merge upstream/master
|
||||
```
|
||||
checkout your feature branch
|
||||
```bash
|
||||
git checkout my-PR-branch
|
||||
git rebase master
|
||||
# follow git instructions to resolve conflists
|
||||
git push -f
|
||||
```
|
||||
|
||||
@@ -2,14 +2,16 @@
|
||||
name: Bug report
|
||||
about: Create a report to help us improve
|
||||
title: ''
|
||||
labels: bug
|
||||
labels: bug, help wanted
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
<!--
|
||||
### Common bugs:
|
||||
1. Tensorboard not showing in Jupyter-notebook see [issue 79](https://github.com/williamFalcon/pytorch-lightning/issues/79).
|
||||
2. PyTorch 1.1.0 vs 1.2.0 support [see FAQ](https://github.com/williamFalcon/pytorch-lightning#faq)
|
||||
1. Tensorboard not showing in Jupyter-notebook see [issue 79](https://github.com/PyTorchLightning/pytorch-lightning/issues/79).
|
||||
2. PyTorch 1.1.0 vs 1.2.0 support [see FAQ](https://github.com/PyTorchLightning/pytorch-lightning#faq)
|
||||
-->
|
||||
|
||||
## 🐛 Bug
|
||||
|
||||
@@ -38,12 +40,12 @@ Minimal means having the shortest code but still preserving the bug. -->
|
||||
### Environment
|
||||
|
||||
Please copy and paste the output from our
|
||||
[environment collection script](https://raw.githubusercontent.com/pytorch/pytorch/master/torch/utils/collect_env.py)
|
||||
[environment collection script](https://raw.githubusercontent.com/PyTorchLightning/pytorch-lightning/master/tests/collect_env_details.py)
|
||||
(or fill out the checklist below manually).
|
||||
|
||||
You can get the script and run it with:
|
||||
```
|
||||
wget https://raw.githubusercontent.com/pytorch/pytorch/master/torch/utils/collect_env.py
|
||||
wget https://raw.githubusercontent.com/PyTorchLightning/pytorch-lightning/master/tests/collect_env_details.py
|
||||
# For security purposes, please check the contents of collect_env.py before running it.
|
||||
python collect_env.py
|
||||
```
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
name: Typos and doc fixes
|
||||
about: Typos and doc fixes
|
||||
title: ''
|
||||
labels: typo
|
||||
labels: typo, documentation
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
# Before submitting
|
||||
|
||||
- [ ] Was this discussed/approved via a Github issue? (no need for typos, doc improvements)
|
||||
- [ ] Did you read the [contributor guideline](https://github.com/williamFalcon/pytorch-lightning/blob/master/.github/CONTRIBUTING.md)?
|
||||
- [ ] Was this discussed/approved via a Github issue? (no need for typos and docs improvements)
|
||||
- [ ] Did you read the [contributor guideline](https://github.com/PyTorchLightning/pytorch-lightning/blob/master/.github/CONTRIBUTING.md), Pull Request section?
|
||||
- [ ] Did you make sure to update the docs?
|
||||
- [ ] Did you write any new necessary tests?
|
||||
- [ ] If you made a notable change (that affects users), did you update the [CHANGELOG](https://github.com/PyTorchLightning/pytorch-lightning/blob/master/CHANGELOG.md)?
|
||||
|
||||
<!-- For CHANGELOG separate each item in unreleased section by blank line to reduce collisions -->
|
||||
|
||||
## What does this PR do?
|
||||
Fixes # (issue).
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
# https://github.com/marketplace/stale
|
||||
|
||||
# Number of days of inactivity before an issue becomes stale
|
||||
daysUntilStale: 60
|
||||
# Number of days of inactivity before a stale issue is closed
|
||||
daysUntilClose: 9
|
||||
# Issues with these labels will never be considered stale
|
||||
exemptLabels:
|
||||
- pinned
|
||||
- security
|
||||
# Label to use when marking an issue as stale
|
||||
staleLabel: wontfix
|
||||
# Comment to post when marking an issue as stale. Set to `false` to disable
|
||||
markComment: >
|
||||
This issue has been automatically marked as stale because it has not had
|
||||
recent activity. It will be closed if no further activity occurs. Thank you
|
||||
for your contributions.
|
||||
# Comment to post when closing a stale issue. Set to `false` to disable
|
||||
closeComment: false
|
||||
@@ -0,0 +1,140 @@
|
||||
name: CI testing
|
||||
|
||||
# see: https://help.github.com/en/actions/reference/events-that-trigger-workflows
|
||||
on:
|
||||
# Trigger the workflow on push or pull request,
|
||||
# but only for the master branch
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
branches:
|
||||
- master
|
||||
|
||||
jobs:
|
||||
build:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
# max-parallel: 6
|
||||
matrix:
|
||||
os: [ubuntu-18.04, windows-2019, macOS-10.15]
|
||||
python-version: [3.6, 3.7, 3.8]
|
||||
requires: ['minimal', 'latest']
|
||||
exclude:
|
||||
# excludes node 4 on macOS
|
||||
- python-version: 3.8
|
||||
requires: 'minimal'
|
||||
|
||||
# Timeout: https://stackoverflow.com/a/59076067/4521646
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- name: Set up Python ${{ matrix.python-version }}
|
||||
uses: actions/setup-python@v1
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
|
||||
# Github Actions: Run step on specific OS: https://stackoverflow.com/a/57948488/4521646
|
||||
- name: Setup macOS
|
||||
if: runner.os == 'macOS'
|
||||
run: |
|
||||
brew install libomp # https://github.com/pytorch/pytorch/issues/20030
|
||||
brew install openmpi # Horovod on macOS requires OpenMPI, Gloo not currently supported
|
||||
|
||||
- name: Setup Windows
|
||||
if: runner.os == 'windows'
|
||||
run: |
|
||||
python -c "lines = [line for line in open('requirements-extra.txt').readlines() if not line.startswith('horovod')] ; open('requirements-extra.txt', 'w').writelines(lines)"
|
||||
|
||||
# TODO: remove after https://github.com/pytorch/pytorch/issues/32186 is resolved
|
||||
- name: Setup Windows on Latest
|
||||
if: runner.os == 'windows' && matrix.requires == 'latest'
|
||||
run: |
|
||||
python -c "req = open('requirements.txt').read().replace('torch>=1.1', 'torch<1.5') ; open('requirements.txt', 'w').write(req)"
|
||||
|
||||
- name: Set min. dependencies
|
||||
if: matrix.requires == 'minimal'
|
||||
run: |
|
||||
python -c "req = open('requirements.txt').read().replace('>', '=') ; open('requirements.txt', 'w').write(req)"
|
||||
python -c "req = open('requirements-extra.txt').read().replace('>', '=') ; open('requirements-extra.txt', 'w').write(req)"
|
||||
|
||||
# Note: This uses an internal pip API and may not always work
|
||||
# https://github.com/actions/cache/blob/master/examples.md#multiple-oss-in-a-workflow
|
||||
- name: Get pip cache
|
||||
id: pip-cache
|
||||
run: |
|
||||
python -c "from pip._internal.locations import USER_CACHE_DIR; print('::set-output name=dir::' + USER_CACHE_DIR)"
|
||||
|
||||
- name: Cache pip
|
||||
uses: actions/cache@v1
|
||||
with:
|
||||
path: ${{ steps.pip-cache.outputs.dir }}
|
||||
key: ${{ runner.os }}-${{ matrix.python-version }}-${{ matrix.requires }}-pip-${{ hashFiles('requirements.txt') }}-${{ hashFiles('requirements-extra.txt') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-${{ matrix.python-version }}-${{ matrix.requires }}-pip-
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
# python -m pip install --upgrade --user pip
|
||||
pip install -r requirements.txt -U -f https://download.pytorch.org/whl/torch_stable.html -q
|
||||
HOROVOD_BUILD_ARCH_FLAGS="-mfma" pip install -r ./tests/requirements-devel.txt -q
|
||||
# pip install tox coverage
|
||||
python --version
|
||||
pip --version
|
||||
pip list
|
||||
shell: bash
|
||||
|
||||
- name: Reinstall Horovod if necessary
|
||||
if: runner.os != 'windows' && matrix.python-version != '3.8'
|
||||
run: |
|
||||
HOROVOD_BUILT=$(python -c "import horovod.torch; horovod.torch.nccl_built(); print('SUCCESS')")
|
||||
if [[ $HOROVOD_BUILT != "SUCCESS" ]]; then
|
||||
pip uninstall -y horovod
|
||||
HOROVOD_BUILD_ARCH_FLAGS="-mfma" pip install --no-cache-dir $(grep "horovod" requirements-extra.txt)
|
||||
fi
|
||||
horovodrun --check-build
|
||||
shell: bash
|
||||
|
||||
- name: Cache datasets
|
||||
uses: actions/cache@v1
|
||||
with:
|
||||
path: tests/Datasets # This path is specific to Ubuntu
|
||||
# Look to see if there is a cache hit for the corresponding requirements file
|
||||
key: mnist-dataset
|
||||
|
||||
- name: Tests
|
||||
# env:
|
||||
# TOXENV: py${{ matrix.python-version }}
|
||||
run: |
|
||||
# tox --sitepackages
|
||||
# flake8 .
|
||||
coverage run --source pytorch_lightning -m py.test pytorch_lightning tests -v --doctest-modules --junitxml=junit/test-results-${{ runner.os }}-${{ matrix.python-version }}-${{ matrix.requires }}.xml
|
||||
coverage report
|
||||
|
||||
- name: Upload pytest test results
|
||||
uses: actions/upload-artifact@master
|
||||
with:
|
||||
name: pytest-results-${{ runner.os }}-${{ matrix.python-version }}-${{ matrix.requires }}
|
||||
path: junit/test-results-${{ runner.os }}-${{ matrix.python-version }}-${{ matrix.requires }}.xml
|
||||
# Use always() to always run this step to publish test results when there are test failures
|
||||
if: always()
|
||||
|
||||
- name: Package Setup
|
||||
run: |
|
||||
check-manifest
|
||||
python setup.py check --metadata --strict
|
||||
python setup.py sdist
|
||||
twine check dist/*
|
||||
#- name: Try install package
|
||||
# if: ! startsWith(matrix.os, 'windows')
|
||||
# run: |
|
||||
# virtualenv vEnv ; source vEnv/bin/activate
|
||||
# pip install --editable . ; cd .. & python -c "import pytorch_lightning ; print(pytorch_lightning.__version__)"
|
||||
# deactivate ; rm -rf vEnv
|
||||
|
||||
- name: Statistics
|
||||
if: success()
|
||||
run: |
|
||||
coverage report
|
||||
@@ -0,0 +1,50 @@
|
||||
name: Publish Docker Releases
|
||||
# https://www.docker.com/blog/first-docker-github-action-is-here
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
release:
|
||||
types:
|
||||
- created
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
python_version: [3.6, 3.7, 3.8]
|
||||
pytorch_version: [1.1, 1.2, 1.3, 1.4, 1.5]
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
|
||||
- name: Publish Master to Docker
|
||||
# publish master
|
||||
uses: docker/build-push-action@v1.1.0
|
||||
if: github.event_name == 'push'
|
||||
with:
|
||||
repository: pytorchlightning/pytorch_lightning
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
dockerfile: docker/Dockerfile
|
||||
buildargs: PYTHON_VERSION=${{ matrix.python_version }},PYTORCH_VERSION=${{ matrix.pytorch_version }}
|
||||
tags: "nightly-py${{ matrix.python_version }}-torch${{ matrix.pytorch_version }}"
|
||||
timeout-minutes: 30
|
||||
|
||||
- name: Get release version
|
||||
if: startsWith(github.ref, 'refs/tags/') || github.event_name == 'release'
|
||||
id: get_version
|
||||
run: echo ::set-env name=RELEASE_VERSION::$(echo ${GITHUB_REF##*/})
|
||||
|
||||
- name: Publish Releases to Docker
|
||||
# only on releases
|
||||
uses: docker/build-push-action@v1.1.0
|
||||
if: startsWith(github.ref, 'refs/tags/') || github.event_name == 'release'
|
||||
with:
|
||||
repository: pytorchlightning/pytorch_lightning
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
dockerfile: docker/Dockerfile
|
||||
buildargs: PYTHON_VERSION=${{ matrix.python_version }},PYTORCH_VERSION=${{ matrix.pytorch_version }},LIGHTNING_VERSION=${{ env.RELEASE_VERSION }}
|
||||
tags: "${{ env.RELEASE_VERSION }}-py${{ matrix.python_version }}-torch${{ matrix.pytorch_version }},latest-py${{ matrix.python_version }}-torch${{ matrix.pytorch_version }}"
|
||||
timeout-minutes: 30
|
||||
@@ -0,0 +1,17 @@
|
||||
name: "Docs check"
|
||||
# https://github.com/marketplace/actions/sphinx-build
|
||||
|
||||
on:
|
||||
- pull_request
|
||||
|
||||
jobs:
|
||||
docs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: ammaraskar/sphinx-action@master
|
||||
with:
|
||||
# git is required to clone the docs theme
|
||||
pre-build-command: "apt-get update -y && apt-get install -y git"
|
||||
docs-folder: "docs/"
|
||||
repo-token: "${{ secrets.GITHUB_TOKEN }}"
|
||||
@@ -0,0 +1,14 @@
|
||||
name: Greetings
|
||||
# https://github.com/marketplace/actions/first-interaction
|
||||
|
||||
on: [issues] # pull_request
|
||||
|
||||
jobs:
|
||||
greeting:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/first-interaction@v1
|
||||
with:
|
||||
repo-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
issue-message: 'Hi! thanks for your contribution!, great first issue!'
|
||||
pr-message: 'Hey thanks for the input! Please give us a bit of time to review it!'
|
||||
@@ -0,0 +1,48 @@
|
||||
name: PyPI Release
|
||||
|
||||
# https://help.github.com/en/actions/reference/events-that-trigger-workflows
|
||||
on:
|
||||
# Trigger the workflow on push or pull request,
|
||||
# but only for the master branch
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
release:
|
||||
types:
|
||||
- created
|
||||
|
||||
# based on https://github.com/pypa/gh-action-pypi-publish
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- name: Set up Python 3.7
|
||||
uses: actions/setup-python@v1
|
||||
with:
|
||||
python-version: 3.7
|
||||
|
||||
- name: Install dependencies
|
||||
run: >-
|
||||
python -m pip install --user --upgrade setuptools wheel
|
||||
- name: Build
|
||||
run: >-
|
||||
python setup.py sdist bdist_wheel
|
||||
|
||||
# We do this, since failures on test.pypi aren't that bad
|
||||
- name: Publish to Test PyPI
|
||||
if: startsWith(github.event.ref, 'refs/tags') || github.event_name == 'release'
|
||||
uses: pypa/gh-action-pypi-publish@master
|
||||
with:
|
||||
user: __token__
|
||||
password: ${{ secrets.test_pypi_password }}
|
||||
repository_url: https://test.pypi.org/legacy/
|
||||
|
||||
- name: Publish distribution 📦 to PyPI
|
||||
if: startsWith(github.event.ref, 'refs/tags') || github.event_name == 'release'
|
||||
uses: pypa/gh-action-pypi-publish@master
|
||||
with:
|
||||
user: __token__
|
||||
password: ${{ secrets.pypi_password }}
|
||||
@@ -0,0 +1,20 @@
|
||||
name: Automatic Rebase
|
||||
# https://github.com/marketplace/actions/automatic-rebase
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
jobs:
|
||||
rebase:
|
||||
name: Rebase
|
||||
if: github.event.issue.pull_request != '' && contains(github.event.comment.body, '/rebase')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Automatic Rebase
|
||||
uses: cirrus-actions/rebase@1.2
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -1,27 +1,27 @@
|
||||
# project
|
||||
.DS_Store
|
||||
.data/
|
||||
run_configs/
|
||||
test_tube_logs/
|
||||
test_tube_data/
|
||||
datasets/
|
||||
model_weights/
|
||||
app/models/
|
||||
pip-wheel-metadata/
|
||||
test_tube_exp/
|
||||
tests/tests_tt_dir/
|
||||
tests/save_dir
|
||||
default/
|
||||
lightning_logs/
|
||||
tests/tests/
|
||||
.vscode/
|
||||
|
||||
# Test-tube
|
||||
test_tube_logs/
|
||||
test_tube_data/
|
||||
test_tube_exp/
|
||||
|
||||
# Documentations
|
||||
docs/source/api
|
||||
docs/source/*.md
|
||||
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*$py.class
|
||||
example.py
|
||||
timit_data/
|
||||
LJSpeech-1.1/
|
||||
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
@@ -30,7 +30,6 @@ LJSpeech-1.1/
|
||||
|
||||
# Distribution / packaging
|
||||
.Python
|
||||
env/
|
||||
ide_layouts/
|
||||
build/
|
||||
develop-eggs/
|
||||
@@ -42,7 +41,6 @@ lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
wheels/
|
||||
*.egg-info/
|
||||
.installed.cfg
|
||||
@@ -68,6 +66,9 @@ nosetests.xml
|
||||
coverage.xml
|
||||
*.cover
|
||||
.hypothesis/
|
||||
tests/tests_tt_dir/
|
||||
tests/save_dir
|
||||
tests/tests/
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
@@ -85,7 +86,7 @@ instance/
|
||||
.scrapy
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
docs/build/
|
||||
|
||||
# PyBuilder
|
||||
target/
|
||||
@@ -107,6 +108,7 @@ celerybeat-schedule
|
||||
|
||||
# virtualenv
|
||||
.venv
|
||||
env/
|
||||
venv/
|
||||
ENV/
|
||||
|
||||
@@ -124,4 +126,11 @@ ENV/
|
||||
.mypy_cache/
|
||||
|
||||
# data
|
||||
.data/
|
||||
datasets/
|
||||
mnist/
|
||||
|
||||
# pl tests
|
||||
ml-runs/
|
||||
*.zip
|
||||
pytorch\ lightning
|
||||
@@ -0,0 +1,49 @@
|
||||
pull_request_rules:
|
||||
|
||||
- name: Automatic merge on approval
|
||||
conditions:
|
||||
- base=master
|
||||
# number of review approvals
|
||||
- "#approved-reviews-by>=3"
|
||||
# no waiting or assigned review
|
||||
- "#review-requested=0"
|
||||
# no requested chnages from any reviewer
|
||||
- "#changes-requested-reviews-by=0"
|
||||
# this serves as ALL check has to pass as we have actually 27 tests in total
|
||||
- "#status-success>=30"
|
||||
# this is just in case since we rely on GPU tests (note: redundand to the above)
|
||||
- status-success=continuous-integration/drone/pr
|
||||
# this is patter-like, unofrunatly serves as `any(...)` (note: redundand to the above)
|
||||
- "status-success~=^ci/circleci:"
|
||||
# no conflict with master branch
|
||||
- -conflict
|
||||
# was not closed yet
|
||||
- -closed
|
||||
actions:
|
||||
delete_head_branch: {}
|
||||
merge:
|
||||
# https://doc.mergify.io/merge-action.html#strict-merge
|
||||
# (on head branch) $ git merge --no-ff base
|
||||
# (on head branch) # Wait for CI to go green
|
||||
# (on head branch) # Squash all commits
|
||||
# (on base branch) $ git merge --ff head
|
||||
strict: true
|
||||
method: squash
|
||||
comment:
|
||||
message: Great job! =)
|
||||
|
||||
- name: warn on conflicts
|
||||
conditions:
|
||||
- conflict
|
||||
actions:
|
||||
comment:
|
||||
message: This pull request is now in conflict... :(
|
||||
|
||||
- name: add core reviewer
|
||||
conditions:
|
||||
# number of review approvals
|
||||
- "#approved-reviews-by<3"
|
||||
actions:
|
||||
request_reviews:
|
||||
teams:
|
||||
- core-contributors
|
||||
@@ -0,0 +1,30 @@
|
||||
# File : .pep8speaks.yml
|
||||
|
||||
scanner:
|
||||
diff_only: True # If False, the entire file touched by the Pull Request is scanned for errors. If True, only the diff is scanned.
|
||||
linter: pycodestyle # Other option is flake8
|
||||
|
||||
pycodestyle: # Same as scanner.linter value. Other option is flake8
|
||||
max-line-length: 110 # Default is 79 in PEP 8
|
||||
ignore: # Errors and warnings to ignore
|
||||
- W504 # line break after binary operator
|
||||
- E402 # module level import not at top of file
|
||||
- E731 # do not assign a lambda expression, use a def
|
||||
- C406 # Unnecessary list literal - rewrite as a dict literal.
|
||||
- E741 # ambiguous variable name
|
||||
- F401
|
||||
- F841
|
||||
|
||||
no_blank_comment: True # If True, no comment is made on PR without any errors.
|
||||
descending_issues_order: False # If True, PEP 8 issues in message will be displayed in descending order of line numbers in the file
|
||||
|
||||
message: # Customize the comment made by the bot,
|
||||
opened: # Messages when a new PR is submitted
|
||||
header: "Hello @{name}! Thanks for opening this PR. "
|
||||
# The keyword {name} is converted into the author's username
|
||||
footer: "Do see the [Hitchhiker's guide to code style](https://goo.gl/hqbW4r)"
|
||||
# The messages can be written as they would over GitHub
|
||||
updated: # Messages when new commits are added to the PR
|
||||
header: "Hello @{name}! Thanks for updating this PR. "
|
||||
footer: "" # Why to comment the link to the style guide everytime? :)
|
||||
no_errors: "There are currently no PEP 8 issues detected in this Pull Request. Cheers! :beers: "
|
||||
@@ -6,8 +6,11 @@
|
||||
version: 2
|
||||
|
||||
# Build documentation in the docs/ directory with Sphinx
|
||||
# reference: https://docs.readthedocs.io/en/stable/config-file/v2.html#sphinx
|
||||
sphinx:
|
||||
configuration: docs/source/conf.py
|
||||
# TODO: set it true and debug failing
|
||||
fail_on_warning: false
|
||||
|
||||
# Build documentation with MkDocs
|
||||
#mkdocs:
|
||||
@@ -20,5 +23,5 @@ formats: all
|
||||
python:
|
||||
version: 3.7
|
||||
install:
|
||||
#- requirements: requirements.txt
|
||||
- requirements: docs/requirements.txt
|
||||
#- requirements: requirements.txt
|
||||
|
||||
@@ -1,9 +1,16 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
# install APEX, see https://github.com/NVIDIA/apex#linux
|
||||
# to imitate SLURM set only single node
|
||||
export SLURM_LOCALID=0
|
||||
|
||||
# use this to run tests
|
||||
rm -rf _ckpt_*
|
||||
rm -rf tests/save_dir*
|
||||
rm -rf tests/mlruns_*
|
||||
rm -rf tests/cometruns*
|
||||
rm -rf tests/tests/*
|
||||
rm -rf lightning_logs
|
||||
coverage run --source pytorch_lightning -m py.test pytorch_lightning tests pl_examples -v --doctest-modules
|
||||
coverage report -m
|
||||
rm -rf ./tests/save_dir*
|
||||
rm -rf ./tests/mlruns_*
|
||||
rm -rf ./tests/cometruns*
|
||||
rm -rf ./tests/wandb*
|
||||
rm -rf ./tests/tests/*
|
||||
rm -rf ./lightning_logs
|
||||
python -m coverage run --source pytorch_lightning -m py.test pytorch_lightning tests pl_examples -v --doctest-modules --flake8
|
||||
python -m coverage report -m
|
||||
|
||||
@@ -1,90 +0,0 @@
|
||||
# vim ft=yaml
|
||||
|
||||
# After changing this file, check it on:
|
||||
# http://yaml-online-parser.appspot.com/
|
||||
|
||||
# See doc/travis_notes.txt for some guidelines
|
||||
|
||||
# this file is *not* meant to cover or endorse the use of travis, but rather to
|
||||
# help confirm pull requests to this project.
|
||||
|
||||
env:
|
||||
global:
|
||||
- DISPLAY=""
|
||||
|
||||
language: python
|
||||
|
||||
matrix:
|
||||
include:
|
||||
- dist: xenial # Ubuntu 16.04
|
||||
python: 3.6
|
||||
env:
|
||||
- TOXENV=py36
|
||||
- MIN_REQUIREMENTS=1
|
||||
- dist: xenial # Ubuntu 16.04
|
||||
python: 3.7
|
||||
env:
|
||||
- TOXENV=py37
|
||||
- MIN_REQUIREMENTS=1
|
||||
- dist: bionic # Ubuntu 18.04
|
||||
python: 3.6
|
||||
env: TOXENV=py36
|
||||
- dist: bionic # Ubuntu 18.04
|
||||
python: 3.7
|
||||
env: TOXENV=py37
|
||||
- os: osx
|
||||
# https://blog.travis-ci.com/2019-08-07-extensive-python-testing-on-travis-ci
|
||||
osx_image: xcode10.3
|
||||
language: generic
|
||||
env: TOXENV=py37
|
||||
#addons:
|
||||
# homebrew:
|
||||
# # update: true
|
||||
# packages: python3.7
|
||||
before_install:
|
||||
- pip3 install virtualenv
|
||||
- virtualenv -p python3 ~/venv
|
||||
- source ~/venv/bin/activate
|
||||
# - os: windows
|
||||
# language: minimal
|
||||
# before_install:
|
||||
# - choco install python3
|
||||
# - export PATH="/c/Python37:/c/Python37/Scripts:$PATH"
|
||||
# env: TOXENV=py37
|
||||
|
||||
# See http://docs.travis-ci.com/user/caching/#pip-cache
|
||||
cache: pip
|
||||
|
||||
install:
|
||||
- pip install future # needed for `builtins`
|
||||
- sudo pip install tox
|
||||
|
||||
before_script:
|
||||
# rewrite all minimal requirements as strict
|
||||
- if [[ "${MIN_REQUIREMENTS}" == "1" ]]; then
|
||||
python -c "req = open('requirements.txt').read().replace('>', '=') ; open('requirements-ci.txt', 'w').write(req)" ;
|
||||
else
|
||||
cp requirements.txt requirements-ci.txt ;
|
||||
fi
|
||||
- pip install -r requirements-ci.txt -U
|
||||
|
||||
script:
|
||||
# integration
|
||||
- tox --sitepackages
|
||||
|
||||
#- python setup.py install --dry-run --user
|
||||
- virtualenv vEnv ;
|
||||
source vEnv/bin/activate
|
||||
- pip install --editable . ;
|
||||
cd .. & python -c "import pytorch_lightning ; print(pytorch_lightning.__version__)"
|
||||
- deactivate ;
|
||||
rm -rf vEnv
|
||||
|
||||
after_success:
|
||||
- coverage report
|
||||
# disable auto coverage bc it isn't accurate since it misses gpu code.
|
||||
# to get coverage, run local and push results
|
||||
# - codecov
|
||||
|
||||
notifications:
|
||||
email: false
|
||||
@@ -0,0 +1,609 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to this project will be documented in this file.
|
||||
|
||||
The format is based on [Keep a Changelog](http://keepachangelog.com/en/1.0.0/).
|
||||
|
||||
## [0.7.6] - 2020-05-16
|
||||
|
||||
### Added
|
||||
|
||||
- Added callback for logging learning rates ([#1498](https://github.com/PyTorchLightning/pytorch-lightning/pull/1498))
|
||||
- Added transfer learning example (for a binary classification task in computer vision) ([#1564](https://github.com/PyTorchLightning/pytorch-lightning/pull/1564))
|
||||
- Added type hints in `Trainer.fit()` and `Trainer.test()` to reflect that also a list of dataloaders can be passed in ([#1723](https://github.com/PyTorchLightning/pytorch-lightning/pull/1723)).
|
||||
- Added auto scaling of batch size ([#1638](https://github.com/PyTorchLightning/pytorch-lightning/pull/1638))
|
||||
- The progress bar metrics now also get updated in `training_epoch_end` ([#1724](https://github.com/PyTorchLightning/pytorch-lightning/pull/1724))
|
||||
- Enable `NeptuneLogger` to work with `distributed_backend=ddp` ([#1753](https://github.com/PyTorchLightning/pytorch-lightning/pull/1753))
|
||||
- Added option to provide seed to random generators to ensure reproducibility ([#1572](https://github.com/PyTorchLightning/pytorch-lightning/pull/1572))
|
||||
- Added override for hparams in `load_from_ckpt` ([#1797](https://github.com/PyTorchLightning/pytorch-lightning/pull/1797))
|
||||
- Added support multi-node distributed execution under `torchelastic` ([#1811](https://github.com/PyTorchLightning/pytorch-lightning/pull/1811), [#1818](https://github.com/PyTorchLightning/pytorch-lightning/pull/1818))
|
||||
- Added using `store_true` for bool args ([#1822](https://github.com/PyTorchLightning/pytorch-lightning/pull/1822), [#1842](https://github.com/PyTorchLightning/pytorch-lightning/pull/1842))
|
||||
- Added dummy logger for internally disabling logging for some features ([#1836](https://github.com/PyTorchLightning/pytorch-lightning/pull/1836))
|
||||
|
||||
### Changed
|
||||
|
||||
- Enable `non-blocking` for device transfers to GPU ([#1843](https://github.com/PyTorchLightning/pytorch-lightning/pull/1843))
|
||||
- Replace mata_tags.csv with hparams.yaml ([#1271](https://github.com/PyTorchLightning/pytorch-lightning/pull/1271))
|
||||
- Reduction when `batch_size < num_gpus` ([#1609](https://github.com/PyTorchLightning/pytorch-lightning/pull/1609))
|
||||
- Updated LightningTemplateModel to look more like Colab example ([#1577](https://github.com/PyTorchLightning/pytorch-lightning/pull/1577))
|
||||
- Don't convert `namedtuple` to `tuple` when transferring the batch to target device ([#1589](https://github.com/PyTorchLightning/pytorch-lightning/pull/1589))
|
||||
- Allow passing hparams as keyword argument to LightningModule when loading from checkpoint ([#1639](https://github.com/PyTorchLightning/pytorch-lightning/pull/1639))
|
||||
- Args should come after the last positional argument ([#1807](https://github.com/PyTorchLightning/pytorch-lightning/pull/1807))
|
||||
|
||||
### Deprecated
|
||||
|
||||
- Deprecated `tags_csv` in favor of `hparams_file` ([#1271](https://github.com/PyTorchLightning/pytorch-lightning/pull/1271))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed broken link in PR template ([#1675](https://github.com/PyTorchLightning/pytorch-lightning/pull/1675))
|
||||
- Fixed ModelCheckpoint not None checking filepath ([#1654](https://github.com/PyTorchLightning/pytorch-lightning/pull/1654))
|
||||
- Trainer now calls `on_load_checkpoint()` when resuming from a checkpoint ([#1666](https://github.com/PyTorchLightning/pytorch-lightning/pull/1666))
|
||||
- Fixed sampler logic for ddp with iterable dataset ([#1734](https://github.com/PyTorchLightning/pytorch-lightning/pull/1734))
|
||||
- Fixed `_reset_eval_dataloader()` for IterableDataset ([#1560](https://github.com/PyTorchLightning/pytorch-lightning/pull/1560))
|
||||
- Fixed Horovod distributed backend to set the `root_gpu` property ([#1669](https://github.com/PyTorchLightning/pytorch-lightning/pull/1669))
|
||||
- Fixed wandb logger `global_step` affects other loggers ([#1492](https://github.com/PyTorchLightning/pytorch-lightning/issues/1485))
|
||||
- Fixed disabling progress bar on non-zero ranks using Horovod backend ([#1709](https://github.com/PyTorchLightning/pytorch-lightning/pull/1709))
|
||||
- Fixed bugs that prevent lr finder to be used together with early stopping and validation dataloaders ([#1676](https://github.com/PyTorchLightning/pytorch-lightning/pull/1676))
|
||||
- Fixed a bug in Trainer that prepended the checkpoint path with `version_` when it shouldn't ([#1748](https://github.com/PyTorchLightning/pytorch-lightning/pull/1748))
|
||||
- Fixed lr key name in case of param groups in LearningRateLogger ([#1719](https://github.com/PyTorchLightning/pytorch-lightning/pull/1719))
|
||||
- Fixed saving native AMP scaler state (introduced in [#1561](https://github.com/PyTorchLightning/pytorch-lightning/pull/1561))
|
||||
- Fixed accumulation parameter and suggestion method for learning rate finder ([#1801](https://github.com/PyTorchLightning/pytorch-lightning/pull/1801))
|
||||
- Fixed num processes wasn't being set properly and auto sampler was ddp failing ([#1819](https://github.com/PyTorchLightning/pytorch-lightning/pull/1819))
|
||||
- Fixed bugs in semantic segmentation example ([#1824](https://github.com/PyTorchLightning/pytorch-lightning/pull/1824))
|
||||
|
||||
## [0.7.5] - 2020-04-27
|
||||
|
||||
### Changed
|
||||
|
||||
- Allow logging of metrics together with `hparams` ([#1630](https://github.com/PyTorchLightning/pytorch-lightning/pull/1630))
|
||||
- Allow metrics logged together with hparams ([#1630](https://github.com/PyTorchLightning/pytorch-lightning/pull/1630))
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed Warning from trainer loop ([#1634](https://github.com/PyTorchLightning/pytorch-lightning/pull/1634))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed ModelCheckpoint not being fixable ([#1632](https://github.com/PyTorchLightning/pytorch-lightning/pull/1632))
|
||||
- Fixed CPU DDP breaking change and DDP change ([#1635](https://github.com/PyTorchLightning/pytorch-lightning/pull/1635))
|
||||
- Tested pickling ([#1636](https://github.com/PyTorchLightning/pytorch-lightning/pull/1636))
|
||||
|
||||
|
||||
## [0.7.4] - 2020-04-26
|
||||
|
||||
### Added
|
||||
|
||||
- Added flag `replace_sampler_ddp` to manually disable sampler replacement in DDP ([#1513](https://github.com/PyTorchLightning/pytorch-lightning/pull/1513))
|
||||
- Added speed parity tests (max 1 sec difference per epoch)([#1482](https://github.com/PyTorchLightning/pytorch-lightning/pull/1482))
|
||||
- Added `auto_select_gpus` flag to trainer that enables automatic selection of available GPUs on exclusive mode systems.
|
||||
- Added learning rate finder ([#1347](https://github.com/PyTorchLightning/pytorch-lightning/pull/1347))
|
||||
- Added support for ddp mode in clusters without SLURM ([#1387](https://github.com/PyTorchLightning/pytorch-lightning/pull/1387))
|
||||
- Added `test_dataloaders` parameter to `Trainer.test()` ([#1434](https://github.com/PyTorchLightning/pytorch-lightning/pull/1434))
|
||||
- Added `terminate_on_nan` flag to trainer that performs a NaN check with each training iteration when set to `True` ([#1475](https://github.com/PyTorchLightning/pytorch-lightning/pull/1475))
|
||||
- Added speed parity tests (max 1 sec difference per epoch)([#1482](https://github.com/PyTorchLightning/pytorch-lightning/pull/1482))
|
||||
- Added `terminate_on_nan` flag to trainer that performs a NaN check with each training iteration when set to `True`. ([#1475](https://github.com/PyTorchLightning/pytorch-lightning/pull/1475))
|
||||
- Added `ddp_cpu` backend for testing ddp without GPUs ([#1158](https://github.com/PyTorchLightning/pytorch-lightning/pull/1158))
|
||||
- Added [Horovod](http://horovod.ai) support as a distributed backend `Trainer(distributed_backend='horovod')` ([#1529](https://github.com/PyTorchLightning/pytorch-lightning/pull/1529))
|
||||
- Added support for 8 core distributed training on Kaggle TPU's ([#1568](https://github.com/PyTorchLightning/pytorch-lightning/pull/1568))
|
||||
- Added support for native AMP ([#1561](https://github.com/PyTorchLightning/pytorch-lightning/pull/1561), [#1580](https://github.com/PyTorchLightning/pytorch-lightning/pull/1580))
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the default behaviour to no longer include a NaN check with each training iteration. ([#1475](https://github.com/PyTorchLightning/pytorch-lightning/pull/1475))
|
||||
- Decoupled the progress bar from trainer` it is a callback now and can be customized or even be replaced entirely ([#1450](https://github.com/PyTorchLightning/pytorch-lightning/pull/1450)).
|
||||
- Changed lr schedule step interval behavior to update every backwards pass instead of every forwards pass ([#1477](https://github.com/PyTorchLightning/pytorch-lightning/pull/1477))
|
||||
- Defines shared proc. rank, remove rank from instances (e.g. loggers) ([#1408](https://github.com/PyTorchLightning/pytorch-lightning/pull/1408))
|
||||
- Updated semantic segmentation example with custom U-Net and logging ([#1371](https://github.com/PyTorchLightning/pytorch-lightning/pull/1371))
|
||||
- Disabled val and test shuffling ([#1600](https://github.com/PyTorchLightning/pytorch-lightning/pull/1600))
|
||||
|
||||
### Deprecated
|
||||
|
||||
- Deprecated `training_tqdm_dict` in favor of `progress_bar_dict` ([#1450](https://github.com/PyTorchLightning/pytorch-lightning/pull/1450)).
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed `test_dataloaders` parameter from `Trainer.fit()` ([#1434](https://github.com/PyTorchLightning/pytorch-lightning/pull/1434))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Added the possibility to pass nested metrics dictionaries to loggers ([#1582](https://github.com/PyTorchLightning/pytorch-lightning/pull/1582))
|
||||
- Fixed memory leak from opt return ([#1528](https://github.com/PyTorchLightning/pytorch-lightning/pull/1528))
|
||||
- Fixed saving checkpoint before deleting old ones ([#1453](https://github.com/PyTorchLightning/pytorch-lightning/pull/1453))
|
||||
- Fixed loggers - flushing last logged metrics even before continue, e.g. `trainer.test()` results ([#1459](https://github.com/PyTorchLightning/pytorch-lightning/pull/1459))
|
||||
- Fixed optimizer configuration when `configure_optimizers` returns dict without `lr_scheduler` ([#1443](https://github.com/PyTorchLightning/pytorch-lightning/pull/1443))
|
||||
- Fixed `LightningModule` - mixing hparams and arguments in `LightningModule.__init__()` crashes load_from_checkpoint() ([#1505](https://github.com/PyTorchLightning/pytorch-lightning/pull/1505))
|
||||
- Added a missing call to the `on_before_zero_grad` model hook ([#1493](https://github.com/PyTorchLightning/pytorch-lightning/pull/1493)).
|
||||
- Allow use of sweeps with `WandbLogger` ([#1512](https://github.com/PyTorchLightning/pytorch-lightning/pull/1512))
|
||||
- Fixed a bug that caused the `callbacks` Trainer argument to reference a global variable ([#1534](https://github.com/PyTorchLightning/pytorch-lightning/pull/1534)).
|
||||
- Fixed a bug that set all boolean CLI arguments from `Trainer.add_argparse_args` always to True ([#1571](https://github.com/PyTorchLightning/pytorch-lightning/pull/1571))
|
||||
- Fixed do not copy the batch when training on a single GPU ([#1576](https://github.com/PyTorchLightning/pytorch-lightning/pull/1576), [#1579](https://github.com/PyTorchLightning/pytorch-lightning/pull/1579))
|
||||
- Fixed soft checkpoint removing on DDP ([#1408](https://github.com/PyTorchLightning/pytorch-lightning/pull/1408))
|
||||
- Fixed automatic parser bug ([#1585](https://github.com/PyTorchLightning/pytorch-lightning/pull/1585))
|
||||
- Fixed bool conversion from string ([#1606](https://github.com/PyTorchLightning/pytorch-lightning/pull/1606))
|
||||
|
||||
## [0.7.3] - 2020-04-09
|
||||
|
||||
### Added
|
||||
|
||||
- Added `rank_zero_warn` for warning only in rank 0 ([#1428](https://github.com/PyTorchLightning/pytorch-lightning/pull/1428))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed default `DistributedSampler` for DDP training ([#1425](https://github.com/PyTorchLightning/pytorch-lightning/pull/1425))
|
||||
- Fixed workers warning not on windows ([#1430](https://github.com/PyTorchLightning/pytorch-lightning/pull/1430))
|
||||
- Fixed returning tuple from `run_training_batch` ([#1431](https://github.com/PyTorchLightning/pytorch-lightning/pull/1431))
|
||||
- Fixed gradient clipping ([#1438](https://github.com/PyTorchLightning/pytorch-lightning/pull/1438))
|
||||
- Fixed pretty print ([#1441](https://github.com/PyTorchLightning/pytorch-lightning/pull/1441))
|
||||
|
||||
|
||||
## [0.7.2] - 2020-04-07
|
||||
|
||||
### Added
|
||||
|
||||
- Added same step loggers' metrics aggregation ([#1278](https://github.com/PyTorchLightning/pytorch-lightning/pull/1278))
|
||||
- Added parity test between a vanilla MNIST model and lightning model ([#1284](https://github.com/PyTorchLightning/pytorch-lightning/pull/1284))
|
||||
- Added parity test between a vanilla RNN model and lightning model ([#1351](https://github.com/PyTorchLightning/pytorch-lightning/pull/1351))
|
||||
- Added Reinforcement Learning - Deep Q-network (DQN) lightning example ([#1232](https://github.com/PyTorchLightning/pytorch-lightning/pull/1232))
|
||||
- Added support for hierarchical `dict` ([#1152](https://github.com/PyTorchLightning/pytorch-lightning/pull/1152))
|
||||
- Added `TrainsLogger` class ([#1122](https://github.com/PyTorchLightning/pytorch-lightning/pull/1122))
|
||||
- Added type hints to `pytorch_lightning.core` ([#946](https://github.com/PyTorchLightning/pytorch-lightning/pull/946))
|
||||
- Added support for `IterableDataset` in validation and testing ([#1104](https://github.com/PyTorchLightning/pytorch-lightning/pull/1104))
|
||||
- Added support for non-primitive types in `hparams` for `TensorboardLogger` ([#1130](https://github.com/PyTorchLightning/pytorch-lightning/pull/1130))
|
||||
- Added a check that stops the training when loss or weights contain `NaN` or `inf` values. ([#1097](https://github.com/PyTorchLightning/pytorch-lightning/pull/1097))
|
||||
- Added support for `IterableDataset` when `val_check_interval=1.0` (default), this will trigger validation at the end of each epoch. ([#1283](https://github.com/PyTorchLightning/pytorch-lightning/pull/1283))
|
||||
- Added `summary` method to Profilers. ([#1259](https://github.com/PyTorchLightning/pytorch-lightning/pull/1259))
|
||||
- Added informative errors if user defined dataloader has zero length ([#1280](https://github.com/PyTorchLightning/pytorch-lightning/pull/1280))
|
||||
- Added testing for python 3.8 ([#915](https://github.com/PyTorchLightning/pytorch-lightning/pull/915))
|
||||
- Added a `training_epoch_end` method which is the mirror of `validation_epoch_end`. ([#1357](https://github.com/PyTorchLightning/pytorch-lightning/pull/1357))
|
||||
- Added model configuration checking ([#1199](https://github.com/PyTorchLightning/pytorch-lightning/pull/1199))
|
||||
- Added support for optimizer frequencies through `LightningModule.configure_optimizers()` ([#1269](https://github.com/PyTorchLightning/pytorch-lightning/pull/1269))
|
||||
- Added option to run without an optimizer by returning `None` from `configure_optimizers`. ([#1279](https://github.com/PyTorchLightning/pytorch-lightning/pull/1279))
|
||||
- Added a warning when the number of data loader workers is small. ([#1378](https://github.com/PyTorchLightning/pytorch-lightning/pull/1378))
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed (renamed and refatored) `TensorRunningMean` -> `TensorRunningAccum`: running accumulations were generalized. ([#1278](https://github.com/PyTorchLightning/pytorch-lightning/pull/1278))
|
||||
- Changed `progress_bar_refresh_rate` trainer flag to disable progress bar when set to 0. ([#1108](https://github.com/PyTorchLightning/pytorch-lightning/pull/1108))
|
||||
- Enhanced `load_from_checkpoint` to also forward params to the model ([#1307](https://github.com/PyTorchLightning/pytorch-lightning/pull/1307))
|
||||
- Updated references to `self.forward()` to instead use the `__call__` interface. ([#1211](https://github.com/PyTorchLightning/pytorch-lightning/pull/1211))
|
||||
- Changed default behaviour of `configure_optimizers` to use no optimizer rather than Adam. ([#1279](https://github.com/PyTorchLightning/pytorch-lightning/pull/1279))
|
||||
- Allow to upload models on W&B ([#1339](https://github.com/PyTorchLightning/pytorch-lightning/pull/1339))
|
||||
- On DP and DDP2 unsqueeze is automated now ([#1319](https://github.com/PyTorchLightning/pytorch-lightning/pull/1319))
|
||||
- Did not always create a DataLoader during reinstantiation, but the same type as before (if subclass of DataLoader) ([#1346](https://github.com/PyTorchLightning/pytorch-lightning/pull/1346))
|
||||
- Did not interfere with a default sampler ([#1318](https://github.com/PyTorchLightning/pytorch-lightning/pull/1318))
|
||||
- Remove default Adam optimizer ([#1317](https://github.com/PyTorchLightning/pytorch-lightning/pull/1317))
|
||||
- Give warnings for unimplemented required lightning methods ([#1317](https://github.com/PyTorchLightning/pytorch-lightning/pull/1317))
|
||||
- Made `evaluate` method private >> `Trainer._evaluate(...)`. ([#1260](https://github.com/PyTorchLightning/pytorch-lightning/pull/1260))
|
||||
- Simplify the PL examples structure (shallower and more readable) ([#1247](https://github.com/PyTorchLightning/pytorch-lightning/pull/1247))
|
||||
- Changed min max gpu memory to be on their own plots ([#1358](https://github.com/PyTorchLightning/pytorch-lightning/pull/1358))
|
||||
- Remove `.item` which causes sync issues ([#1254](https://github.com/PyTorchLightning/pytorch-lightning/pull/1254))
|
||||
- Changed smoothing in TQDM to decrease variability of time remaining between training / eval ([#1194](https://github.com/PyTorchLightning/pytorch-lightning/pull/1194))
|
||||
- Change default logger to dedicated one ([#1064](https://github.com/PyTorchLightning/pytorch-lightning/pull/1064))
|
||||
|
||||
### Deprecated
|
||||
|
||||
- Deprecated Trainer argument `print_nan_grads` ([#1097](https://github.com/PyTorchLightning/pytorch-lightning/pull/1097))
|
||||
- Deprecated Trainer argument `show_progress_bar` ([#1108](https://github.com/PyTorchLightning/pytorch-lightning/pull/1108))
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed test for no test dataloader in .fit ([#1495](https://github.com/PyTorchLightning/pytorch-lightning/pull/1495))
|
||||
- Removed duplicated module `pytorch_lightning.utilities.arg_parse` for loading CLI arguments ([#1167](https://github.com/PyTorchLightning/pytorch-lightning/pull/1167))
|
||||
- Removed wandb logger's `finalize` method ([#1193](https://github.com/PyTorchLightning/pytorch-lightning/pull/1193))
|
||||
- Dropped `torchvision` dependency in tests and added own MNIST dataset class instead ([#986](https://github.com/PyTorchLightning/pytorch-lightning/pull/986))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `model_checkpoint` when saving all models ([#1359](https://github.com/PyTorchLightning/pytorch-lightning/pull/1359))
|
||||
- `Trainer.add_argparse_args` classmethod fixed. Now it adds a type for the arguments ([#1147](https://github.com/PyTorchLightning/pytorch-lightning/pull/1147))
|
||||
- Fixed bug related to type checking of `ReduceLROnPlateau` lr schedulers([#1126](https://github.com/PyTorchLightning/pytorch-lightning/pull/1126))
|
||||
- Fixed a bug to ensure lightning checkpoints to be backward compatible ([#1132](https://github.com/PyTorchLightning/pytorch-lightning/pull/1132))
|
||||
- Fixed a bug that created an extra dataloader with active `reload_dataloaders_every_epoch` ([#1196](https://github.com/PyTorchLightning/pytorch-lightning/pull/1196))
|
||||
- Fixed all warnings and errors in the docs build process ([#1191](https://github.com/PyTorchLightning/pytorch-lightning/pull/1191))
|
||||
- Fixed an issue where `val_percent_check=0` would not disable validation ([#1251](https://github.com/PyTorchLightning/pytorch-lightning/pull/1251))
|
||||
- Fixed average of incomplete `TensorRunningMean` ([#1309](https://github.com/PyTorchLightning/pytorch-lightning/pull/1309))
|
||||
- Fixed `WandbLogger.watch` with `wandb.init()` ([#1311](https://github.com/PyTorchLightning/pytorch-lightning/pull/1311))
|
||||
- Fixed an issue with early stopping that would prevent it from monitoring training metrics when validation is disabled / not implemented ([#1235](https://github.com/PyTorchLightning/pytorch-lightning/pull/1235)).
|
||||
- Fixed a bug that would cause `trainer.test()` to run on the validation set when overloading `validation_epoch_end` and `test_end` ([#1353](https://github.com/PyTorchLightning/pytorch-lightning/pull/1353))
|
||||
- Fixed `WandbLogger.watch` - use of the watch method without importing `wandb` ([#1311](https://github.com/PyTorchLightning/pytorch-lightning/pull/1311))
|
||||
- Fixed `WandbLogger` to be used with 'ddp' - allow reinits in sub-processes ([#1149](https://github.com/PyTorchLightning/pytorch-lightning/pull/1149), [#1360](https://github.com/PyTorchLightning/pytorch-lightning/pull/1360))
|
||||
- Made `training_epoch_end` behave like `validation_epoch_end` ([#1357](https://github.com/PyTorchLightning/pytorch-lightning/pull/1357))
|
||||
- Fixed `fast_dev_run` running validation twice ([#1365](https://github.com/PyTorchLightning/pytorch-lightning/pull/1365))
|
||||
- Fixed pickle error from quick patch `__code__` ([#1352](https://github.com/PyTorchLightning/pytorch-lightning/pull/1352))
|
||||
- Fixed memory leak on GPU0 ([#1094](https://github.com/PyTorchLightning/pytorch-lightning/pull/1094), [#1349](https://github.com/PyTorchLightning/pytorch-lightning/pull/1349))
|
||||
- Fixed checkpointing interval ([#1272](https://github.com/PyTorchLightning/pytorch-lightning/pull/1272))
|
||||
- Fixed validation and training loops run the partial dataset ([#1192](https://github.com/PyTorchLightning/pytorch-lightning/pull/1192))
|
||||
- Fixed running `on_validation_end` only on main process in DDP ([#1125](https://github.com/PyTorchLightning/pytorch-lightning/pull/1125))
|
||||
- Fixed `load_spawn_weights` only in proc rank 0 ([#1385](https://github.com/PyTorchLightning/pytorch-lightning/pull/1385))
|
||||
- Fixes `use_amp` issue ([#1145](https://github.com/PyTorchLightning/pytorch-lightning/pull/1145))
|
||||
- Fixes using deprecated `use_amp` attribute ([#1145](https://github.com/PyTorchLightning/pytorch-lightning/pull/1145))
|
||||
- Fixed Tensorboard logger error: lightning_logs directory not exists in multi-node DDP on nodes with rank != 0 ([#1377](https://github.com/PyTorchLightning/pytorch-lightning/pull/1377))
|
||||
- Fixed `Unimplemented backend XLA` error on TPU ([#1387](https://github.com/PyTorchLightning/pytorch-lightning/pull/1387))
|
||||
|
||||
## [0.7.1] - 2020-03-07
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixes `print` issues and `data_loader` ([#1080](https://github.com/PyTorchLightning/pytorch-lightning/pull/1080))
|
||||
|
||||
## [0.7.0] - 2020-03-06
|
||||
|
||||
### Added
|
||||
|
||||
- Added automatic sampler setup. Depending on DDP or TPU, lightning configures the sampler correctly (user needs to do nothing) ([#926](https://github.com/PyTorchLightning/pytorch-lightning/pull/926))
|
||||
- Added `reload_dataloaders_every_epoch=False` flag for trainer. Some users require reloading data every epoch ([#926](https://github.com/PyTorchLightning/pytorch-lightning/pull/926))
|
||||
- Added `progress_bar_refresh_rate=50` flag for trainer. Throttle refresh rate on notebooks ([#926](https://github.com/PyTorchLightning/pytorch-lightning/pull/926))
|
||||
- Updated governance docs
|
||||
- Added a check to ensure that the metric used for early stopping exists before training commences ([#542](https://github.com/PyTorchLightning/pytorch-lightning/pull/542))
|
||||
- Added `optimizer_idx` argument to `backward` hook ([#733](https://github.com/PyTorchLightning/pytorch-lightning/pull/733))
|
||||
- Added `entity` argument to `WandbLogger` to be passed to `wandb.init` ([#783](https://github.com/PyTorchLightning/pytorch-lightning/pull/783))
|
||||
- Added a tool for profiling training runs ([#782](https://github.com/PyTorchLightning/pytorch-lightning/pull/782))
|
||||
- Improved flexibility for naming of TensorBoard logs, can now set `version` to a `str` to just save to that directory, and use `name=''` to prevent experiment-name directory ([#804](https://github.com/PyTorchLightning/pytorch-lightning/pull/804))
|
||||
- Added option to specify `step` key when logging metrics ([#808](https://github.com/PyTorchLightning/pytorch-lightning/pull/808))
|
||||
- Added `train_dataloader`, `val_dataloader` and `test_dataloader` arguments to `Trainer.fit()`, for alternative data parsing ([#759](https://github.com/PyTorchLightning/pytorch-lightning/pull/759))
|
||||
- Added Tensor Processing Unit (TPU) support ([#868](https://github.com/PyTorchLightning/pytorch-lightning/pull/868))
|
||||
- Added semantic segmentation example ([#751](https://github.com/PyTorchLightning/pytorch-lightning/pull/751),[#876](https://github.com/PyTorchLightning/pytorch-lightning/pull/876), [#881](https://github.com/PyTorchLightning/pytorch-lightning/pull/881))
|
||||
- Split callbacks in multiple files ([#849](https://github.com/PyTorchLightning/pytorch-lightning/pull/849))
|
||||
- Support for user defined callbacks ([#889](https://github.com/PyTorchLightning/pytorch-lightning/pull/889) and [#950](https://github.com/PyTorchLightning/pytorch-lightning/pull/950))
|
||||
- Added support for multiple loggers to be passed to `Trainer` as an iterable (e.g. list, tuple, etc.) ([#903](https://github.com/PyTorchLightning/pytorch-lightning/pull/903))
|
||||
- Added support for step-based learning rate scheduling ([#941](https://github.com/PyTorchLightning/pytorch-lightning/pull/941))
|
||||
- Added support for logging `hparams` as dict ([#1029](https://github.com/PyTorchLightning/pytorch-lightning/pull/1029))
|
||||
- Checkpoint and early stopping now work without val. step ([#1041](https://github.com/PyTorchLightning/pytorch-lightning/pull/1041))
|
||||
- Support graceful training cleanup after Keyboard Interrupt ([#856](https://github.com/PyTorchLightning/pytorch-lightning/pull/856), [#1019](https://github.com/PyTorchLightning/pytorch-lightning/pull/1019))
|
||||
- Added type hints for function arguments ([#912](https://github.com/PyTorchLightning/pytorch-lightning/pull/912), )
|
||||
- Added default `argparser` for `Trainer` ([#952](https://github.com/PyTorchLightning/pytorch-lightning/pull/1023), [#1023](https://github.com/PyTorchLightning/pytorch-lightning/pull/1023))
|
||||
- Added TPU gradient clipping ([#963](https://github.com/PyTorchLightning/pytorch-lightning/pull/963))
|
||||
- Added max/min number of steps in `Trainer` ([#728](https://github.com/PyTorchLightning/pytorch-lightning/pull/728))
|
||||
|
||||
### Changed
|
||||
|
||||
- Improved `NeptuneLogger` by adding `close_after_fit` argument to allow logging after training([#908](https://github.com/PyTorchLightning/pytorch-lightning/pull/1084))
|
||||
- Changed default TQDM to use `tqdm.auto` for prettier outputs in IPython notebooks ([#752](https://github.com/PyTorchLightning/pytorch-lightning/pull/752))
|
||||
- Changed `pytorch_lightning.logging` to `pytorch_lightning.loggers` ([#767](https://github.com/PyTorchLightning/pytorch-lightning/pull/767))
|
||||
- Moved the default `tqdm_dict` definition from Trainer to `LightningModule`, so it can be overridden by the user ([#749](https://github.com/PyTorchLightning/pytorch-lightning/pull/749))
|
||||
- Moved functionality of `LightningModule.load_from_metrics` into `LightningModule.load_from_checkpoint` ([#995](https://github.com/PyTorchLightning/pytorch-lightning/pull/995))
|
||||
- Changed Checkpoint path parameter from `filepath` to `dirpath` ([#1016](https://github.com/PyTorchLightning/pytorch-lightning/pull/1016))
|
||||
- Freezed models `hparams` as `Namespace` property ([#1029](https://github.com/PyTorchLightning/pytorch-lightning/pull/1029))
|
||||
- Dropped `logging` config in package init ([#1015](https://github.com/PyTorchLightning/pytorch-lightning/pull/1015))
|
||||
- Renames model steps ([#1051](https://github.com/PyTorchLightning/pytorch-lightning/pull/1051))
|
||||
- `training_end` >> `training_epoch_end`
|
||||
- `validation_end` >> `validation_epoch_end`
|
||||
- `test_end` >> `test_epoch_end`
|
||||
- Refactor dataloading, supports infinite dataloader ([#955](https://github.com/PyTorchLightning/pytorch-lightning/pull/955))
|
||||
- Create single file in `TensorBoardLogger` ([#777](https://github.com/PyTorchLightning/pytorch-lightning/pull/777))
|
||||
|
||||
### Deprecated
|
||||
|
||||
- Deprecated `pytorch_lightning.logging` ([#767](https://github.com/PyTorchLightning/pytorch-lightning/pull/767))
|
||||
- Deprecated `LightningModule.load_from_metrics` in favour of `LightningModule.load_from_checkpoint` ([#995](https://github.com/PyTorchLightning/pytorch-lightning/pull/995), [#1079](https://github.com/PyTorchLightning/pytorch-lightning/pull/1079))
|
||||
- Deprecated `@data_loader` decorator ([#926](https://github.com/PyTorchLightning/pytorch-lightning/pull/926))
|
||||
- Deprecated model steps `training_end`, `validation_end` and `test_end` ([#1051](https://github.com/PyTorchLightning/pytorch-lightning/pull/1051), [#1056](https://github.com/PyTorchLightning/pytorch-lightning/pull/1056))
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed dependency on `pandas` ([#736](https://github.com/PyTorchLightning/pytorch-lightning/pull/736))
|
||||
- Removed dependency on `torchvision` ([#797](https://github.com/PyTorchLightning/pytorch-lightning/pull/797))
|
||||
- Removed dependency on `scikit-learn` ([#801](https://github.com/PyTorchLightning/pytorch-lightning/pull/801))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where early stopping `on_end_epoch` would be called inconsistently when `check_val_every_n_epoch == 0` ([#743](https://github.com/PyTorchLightning/pytorch-lightning/pull/743))
|
||||
- Fixed a bug where the model checkpointer didn't write to the same directory as the logger ([#771](https://github.com/PyTorchLightning/pytorch-lightning/pull/771))
|
||||
- Fixed a bug where the `TensorBoardLogger` class would create an additional empty log file during fitting ([#777](https://github.com/PyTorchLightning/pytorch-lightning/pull/777))
|
||||
- Fixed a bug where `global_step` was advanced incorrectly when using `accumulate_grad_batches > 1` ([#832](https://github.com/PyTorchLightning/pytorch-lightning/pull/832))
|
||||
- Fixed a bug when calling `self.logger.experiment` with multiple loggers ([#1009](https://github.com/PyTorchLightning/pytorch-lightning/pull/1009))
|
||||
- Fixed a bug when calling `logger.append_tags` on a `NeptuneLogger` with a single tag ([#1009](https://github.com/PyTorchLightning/pytorch-lightning/pull/1009))
|
||||
- Fixed sending back data from `.spawn` by saving and loading the trained model in/out of the process ([#1017](https://github.com/PyTorchLightning/pytorch-lightning/pull/1017)
|
||||
- Fixed port collision on DDP ([#1010](https://github.com/PyTorchLightning/pytorch-lightning/pull/1010))
|
||||
- Fixed/tested pass overrides ([#918](https://github.com/PyTorchLightning/pytorch-lightning/pull/918))
|
||||
- Fixed comet logger to log after train ([#892](https://github.com/PyTorchLightning/pytorch-lightning/pull/892))
|
||||
- Remove deprecated args to learning rate step function ([#890](https://github.com/PyTorchLightning/pytorch-lightning/pull/890))
|
||||
|
||||
## [0.6.0] - 2020-01-21
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for resuming from a specific checkpoint via `resume_from_checkpoint` argument ([#516](https://github.com/PyTorchLightning/pytorch-lightning/pull/516))
|
||||
- Added support for `ReduceLROnPlateau` scheduler ([#320](https://github.com/PyTorchLightning/pytorch-lightning/pull/320))
|
||||
- Added support for Apex mode `O2` in conjunction with Data Parallel ([#493](https://github.com/PyTorchLightning/pytorch-lightning/pull/493))
|
||||
- Added option (`save_top_k`) to save the top k models in the `ModelCheckpoint` class ([#128](https://github.com/PyTorchLightning/pytorch-lightning/pull/128))
|
||||
- Added `on_train_start` and `on_train_end` hooks to `ModelHooks` ([#598](https://github.com/PyTorchLightning/pytorch-lightning/pull/598))
|
||||
- Added `TensorBoardLogger` ([#607](https://github.com/PyTorchLightning/pytorch-lightning/pull/607))
|
||||
- Added support for weight summary of model with multiple inputs ([#543](https://github.com/PyTorchLightning/pytorch-lightning/pull/543))
|
||||
- Added `map_location` argument to `load_from_metrics` and `load_from_checkpoint` ([#625](https://github.com/PyTorchLightning/pytorch-lightning/pull/625))
|
||||
- Added option to disable validation by setting `val_percent_check=0` ([#649](https://github.com/PyTorchLightning/pytorch-lightning/pull/649))
|
||||
- Added `NeptuneLogger` class ([#648](https://github.com/PyTorchLightning/pytorch-lightning/pull/648))
|
||||
- Added `WandbLogger` class ([#627](https://github.com/PyTorchLightning/pytorch-lightning/pull/627))
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the default progress bar to print to stdout instead of stderr ([#531](https://github.com/PyTorchLightning/pytorch-lightning/pull/531))
|
||||
- Renamed `step_idx` to `step`, `epoch_idx` to `epoch`, `max_num_epochs` to `max_epochs` and `min_num_epochs` to `min_epochs` ([#589](https://github.com/PyTorchLightning/pytorch-lightning/pull/589))
|
||||
- Renamed `total_batch_nb` to `total_batches`, `nb_val_batches` to `num_val_batches`, `nb_training_batches` to `num_training_batches`, `max_nb_epochs` to `max_epochs`, `min_nb_epochs` to `min_epochs`, `nb_test_batches` to `num_test_batches`, and `nb_val_batches` to `num_val_batches` ([#567](https://github.com/PyTorchLightning/pytorch-lightning/pull/567))
|
||||
- Changed gradient logging to use parameter names instead of indexes ([#660](https://github.com/PyTorchLightning/pytorch-lightning/pull/660))
|
||||
- Changed the default logger to `TensorBoardLogger` ([#609](https://github.com/PyTorchLightning/pytorch-lightning/pull/609))
|
||||
- Changed the directory for tensorboard logging to be the same as model checkpointing ([#706](https://github.com/PyTorchLightning/pytorch-lightning/pull/706))
|
||||
|
||||
### Deprecated
|
||||
|
||||
- Deprecated `max_nb_epochs` and `min_nb_epochs` ([#567](https://github.com/PyTorchLightning/pytorch-lightning/pull/567))
|
||||
- Deprecated the `on_sanity_check_start` hook in `ModelHooks` ([#598](https://github.com/PyTorchLightning/pytorch-lightning/pull/598))
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the `save_best_only` argument from `ModelCheckpoint`, use `save_top_k=1` instead ([#128](https://github.com/PyTorchLightning/pytorch-lightning/pull/128))
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug which ocurred when using Adagrad with cuda ([#554](https://github.com/PyTorchLightning/pytorch-lightning/pull/554))
|
||||
- Fixed a bug where training would be on the GPU despite setting `gpus=0` or `gpus=[]` ([#561](https://github.com/PyTorchLightning/pytorch-lightning/pull/561))
|
||||
- Fixed an error with `print_nan_gradients` when some parameters do not require gradient ([#579](https://github.com/PyTorchLightning/pytorch-lightning/pull/579))
|
||||
- Fixed a bug where the progress bar would show an incorrect number of total steps during the validation sanity check when using multiple validation data loaders ([#597](https://github.com/PyTorchLightning/pytorch-lightning/pull/597))
|
||||
- Fixed support for PyTorch 1.1.0 ([#552](https://github.com/PyTorchLightning/pytorch-lightning/pull/552))
|
||||
- Fixed an issue with early stopping when using a `val_check_interval < 1.0` in `Trainer` ([#492](https://github.com/PyTorchLightning/pytorch-lightning/pull/492))
|
||||
- Fixed bugs relating to the `CometLogger` object that would cause it to not work properly ([#481](https://github.com/PyTorchLightning/pytorch-lightning/pull/481))
|
||||
- Fixed a bug that would occur when returning `-1` from `on_batch_start` following an early exit or when the batch was `None` ([#509](https://github.com/PyTorchLightning/pytorch-lightning/pull/509))
|
||||
- Fixed a potential race condition with several processes trying to create checkpoint directories ([#530](https://github.com/PyTorchLightning/pytorch-lightning/pull/530))
|
||||
- Fixed a bug where batch 'segments' would remain on the GPU when using `truncated_bptt > 1` ([#532](https://github.com/PyTorchLightning/pytorch-lightning/pull/532))
|
||||
- Fixed a bug when using `IterableDataset` ([#547](https://github.com/PyTorchLightning/pytorch-lightning/pull/547))
|
||||
- Fixed a bug where `.item` was called on non-tensor objects ([#602](https://github.com/PyTorchLightning/pytorch-lightning/pull/602))
|
||||
- Fixed a bug where `Trainer.train` would crash on an uninitialized variable if the trainer was run after resuming from a checkpoint that was already at `max_epochs` ([#608](https://github.com/PyTorchLightning/pytorch-lightning/pull/608))
|
||||
- Fixed a bug where early stopping would begin two epochs early ([#617](https://github.com/PyTorchLightning/pytorch-lightning/pull/617))
|
||||
- Fixed a bug where `num_training_batches` and `num_test_batches` would sometimes be rounded down to zero ([#649](https://github.com/PyTorchLightning/pytorch-lightning/pull/649))
|
||||
- Fixed a bug where an additional batch would be processed when manually setting `num_training_batches` ([#653](https://github.com/PyTorchLightning/pytorch-lightning/pull/653))
|
||||
- Fixed a bug when batches did not have a `.copy` method ([#701](https://github.com/PyTorchLightning/pytorch-lightning/pull/701))
|
||||
- Fixed a bug when using `log_gpu_memory=True` in Python 3.6 ([#715](https://github.com/PyTorchLightning/pytorch-lightning/pull/715))
|
||||
- Fixed a bug where checkpoint writing could exit before completion, giving incomplete checkpoints ([#689](https://github.com/PyTorchLightning/pytorch-lightning/pull/689))
|
||||
- Fixed a bug where `on_train_end` was not called when ealy stopping ([#723](https://github.com/PyTorchLightning/pytorch-lightning/pull/723))
|
||||
|
||||
## [0.5.3] - 2019-11-06
|
||||
|
||||
### Added
|
||||
|
||||
- Added option to disable default logger, checkpointer, and early stopping by passing `logger=False`, `checkpoint_callback=False` and `early_stop_callback=False` respectively
|
||||
- Added `CometLogger` for use with Comet.ml
|
||||
- Added `val_check_interval` argument to `Trainer` allowing validition to be performed at every given number of batches
|
||||
- Added functionality to save and load hyperparameters using the standard checkpoint mechanism
|
||||
- Added call to `torch.cuda.empty_cache` before training starts
|
||||
- Added option for user to override the call t `backward`
|
||||
- Added support for truncated backprop through time via the `truncated_bptt_steps` argument in `Trainer`
|
||||
- Added option to operate on all outputs from `training_step` in DDP2
|
||||
- Added a hook for modifying DDP init
|
||||
- Added a hook for modifying Apex
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed experiment version to be padded with zeros (e.g. `/dir/version_9` becomes `/dir/version_0009`)
|
||||
- Changed callback metrics to include any metrics given in logs or progress bar
|
||||
- Changed the default for `save_best_only` in `ModelCheckpoint` to `True`
|
||||
- Added `tng_data_loader` for backwards compatibility
|
||||
- Renamed `MLFlowLogger.client` to `MLFlowLogger.experiment` for consistency
|
||||
- Moved `global_step` increment to happen after the batch has been processed
|
||||
- Changed weights restore to first attempt HPC weights before restoring normally, preventing both weights being restored and running out of memory
|
||||
- Changed progress bar functionality to add multiple progress bars for train/val/test
|
||||
- Changed calls to `print` to use `logging` instead
|
||||
|
||||
### Deprecated
|
||||
|
||||
- Deprecated `tng_dataloader`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed an issue where the number of batches was off by one during training
|
||||
- Fixed a bug that occured when setting a ckeckpoint callback and `early_stop_callback=False`
|
||||
- Fixed an error when importing CometLogger
|
||||
- Fixed a bug where the `gpus` argument had some unexpected behaviour
|
||||
- Fixed a bug where the computed total number of batches was sometimes incorrect
|
||||
- Fixed a bug where the progress bar would sometimes not show the total number of batches in test mode
|
||||
- Fixed a bug when using the `log_gpu_memory='min_max'` option in `Trainer`
|
||||
- Fixed a bug where checkpointing would sometimes erase the current directory
|
||||
|
||||
## [0.5.2] - 2019-10-10
|
||||
|
||||
### Added
|
||||
|
||||
- Added `weights_summary` argument to `Trainer` to be set to `full` (full summary), `top` (just top level modules) or other
|
||||
- Added `tags` argument to `MLFlowLogger`
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed default for `amp_level` to `O1`
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the `print_weights_summary` argument from `Trainer`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where logs were not written properly
|
||||
- Fixed a bug where `logger.finalize` wasn't called after training is complete
|
||||
- Fixed callback metric errors in DDP
|
||||
- Fixed a bug where `TestTubeLogger` didn't log to the correct directory
|
||||
|
||||
## [0.5.1] - 2019-10-05
|
||||
|
||||
### Added
|
||||
|
||||
- Added the `LightningLoggerBase` class for experiment loggers
|
||||
- Added `MLFlowLogger` for logging with `mlflow`
|
||||
- Added `TestTubeLogger` for logging with `test_tube`
|
||||
- Added a different implementation of DDP (`distributed_backed='ddp2'`) where every node has one model using all GPUs
|
||||
- Added support for optimisers which require a closure (e.g. LBFGS)
|
||||
- Added automatic `MASTER_PORT` defualt for DDP when not set manually
|
||||
- Added new GPU memory logging options `'min_max'` (log only the min/max utilization) and `'all'` (log all the GPU memory)
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed schedulers to always be called with the current epoch
|
||||
- Changed `test_tube` to an optional dependency
|
||||
- Changed data loaders to internally use a getter instead of a python property
|
||||
- Disabled auto GPU loading when restoring weights to prevent out of memory errors
|
||||
- Changed logging, early stopping and checkpointing to occur by default
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug with samplers that do not specify `set_epoch`
|
||||
- Fixed a bug when using the `MLFlowLogger` with unsupported data types, this will now raise a warning
|
||||
- Fixed a bug where gradient norms were alwasy zero using `track_grad_norm`
|
||||
- Fixed a bug which causes a crash when logging memory
|
||||
|
||||
## [0.5.0] - 2019-09-26
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed `data_batch` argument to `batch` throughout
|
||||
- Changed `batch_i` argument to `batch_idx` throughout
|
||||
- Changed `tng_dataloader` method to `train_dataloader`
|
||||
- Changed `on_tng_metrics` method to `on_training_metrics`
|
||||
- Changed `gradient_clip` argument to `gradient_clip_val`
|
||||
- Changed `add_log_row_interval` to `row_log_interval`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug with tensorboard logging in multi-gpu setup
|
||||
|
||||
## [0.4.9] - 2019-09-16
|
||||
|
||||
### Added
|
||||
|
||||
- Added the flag `log_gpu_memory` to `Trainer` to deactivate logging of GPU memory utilization
|
||||
- Added SLURM resubmit functionality (port from test-tube)
|
||||
- Added optional weight_save_path to trainer to remove the need for a checkpoint_callback when using cluster training
|
||||
- Added option to use single gpu per node with `DistributedDataParallel`
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed functionality of `validation_end` and `test_end` with multiple dataloaders to be given all of the dataloaders at once rather than in seperate calls
|
||||
- Changed print_nan_grads to only print the parameter value and gradients when they contain NaN
|
||||
- Changed gpu API to take integers as well (e.g. `gpus=2` instead of `gpus=[0, 1]`)
|
||||
- All models now loaded on to CPU to avoid device and out of memory issues in PyTorch
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where data types that implement `.to` but not `.cuda` would not be properly moved onto the GPU
|
||||
- Fixed a bug where data would not be re-shuffled every epoch when using a `DistributedSampler`
|
||||
|
||||
## [0.4.8] - 2019-08-31
|
||||
|
||||
### Added
|
||||
|
||||
- Added `test_step` and `test_end` methods, used when `Trainer.test` is called
|
||||
- Added `GradientAccumulationScheduler` callback which can be used to schedule changes to the number of accumulation batches
|
||||
- Added option to skip the validation sanity check by setting `nb_sanity_val_steps = 0`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug when setting `nb_sanity_val_steps = 0`
|
||||
|
||||
## [0.4.7] - 2019-08-24
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the default `val_check_interval` to `1.0`
|
||||
- Changed defaults for `nb_val_batches`, `nb_tng_batches` and `nb_test_batches` to 0
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where the full validation set as used despite setting `val_percent_check`
|
||||
- Fixed a bug where an `Exception` was thrown when using a data set containing a single batch
|
||||
- Fixed a bug where an `Exception` was thrown if no `val_dataloader` was given
|
||||
- Fixed a bug where tuples were not properly transfered to the GPU
|
||||
- Fixed a bug where data of a non standard type was not properly handled by the trainer
|
||||
- Fixed a bug when loading data as a tuple
|
||||
- Fixed a bug where `AttributeError` could be suppressed by the `Trainer`
|
||||
|
||||
## [0.4.6] - 2019-08-15
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for data to be given as a `dict` or `list` with a single gpu
|
||||
- Added support for `configure_optimizers` to return a single optimizer, two list (optimizers and schedulers), or a single list
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where returning just an optimizer list (i.e. without schedulers) from `configure_optimizers` would throw an `Exception`
|
||||
|
||||
## [0.4.5] - 2019-08-13
|
||||
|
||||
### Added
|
||||
|
||||
- Added `optimizer_step` method that can be overridden to change the standard optimizer behaviour
|
||||
|
||||
## [0.4.4] - 2019-08-12
|
||||
|
||||
### Added
|
||||
|
||||
- Added supoort for multiple validation dataloaders
|
||||
- Added support for latest test-tube logger (optimised for `torch==1.2.0`)
|
||||
|
||||
### Changed
|
||||
|
||||
- `validation_step` and `val_dataloader` are now optional
|
||||
- `lr_scheduler` is now activated after epoch
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where a warning would show when using `lr_scheduler` in `torch>1.1.0`
|
||||
- Fixed a bug where an `Exception` would be thrown if using `torch.DistributedDataParallel` without using a `DistributedSampler`, this now throws a `Warning` instead
|
||||
|
||||
## [0.4.3] - 2019-08-10
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where accumulate gradients would scale the loss incorrectly
|
||||
|
||||
## [0.4.2] - 2019-08-08
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed install requirement to `torch==1.2.0`
|
||||
|
||||
## [0.4.1] - 2019-08-08
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed install requirement to `torch==1.1.0`
|
||||
|
||||
## [0.4.0] - 2019-08-08
|
||||
|
||||
### Added
|
||||
|
||||
- Added 16-bit support for a single GPU
|
||||
- Added support for training continuation (preserves epoch, global step etc.)
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed `training_step` and `validation_step`, outputs will no longer be automatically reduced
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed need for `Experiment` object in `Trainer`
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed issues with reducing outputs from generative models (such as images and text)
|
||||
|
||||
## [0.3.6] - 2019-07-25
|
||||
|
||||
### Added
|
||||
|
||||
- Added a decorator to do lazy data loading internally
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a bug where `Experiment` object was not process safe, potentially causing logs to be overwritten
|
||||
|
||||
## [0.3.5] - 2019-MM-DD
|
||||
|
||||
## [0.3.4] - 2019-MM-DD
|
||||
|
||||
## [0.3.3] - 2019-MM-DD
|
||||
|
||||
## [0.3.2] - 2019-MM-DD
|
||||
|
||||
## [0.3.1] - 2019-MM-DD
|
||||
|
||||
## [0.2.x] - YYYY-MM-DD
|
||||
|
||||
## [0.1.x] - YYYY-MM-DD
|
||||
@@ -186,7 +186,7 @@
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
Copyright 2018-2020 William Falcon
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
# Manifest syntax https://docs.python.org/2/distutils/sourcedist.html
|
||||
graft wheelhouse
|
||||
|
||||
recursive-include birl *.py
|
||||
recursive-exclude __pycache__ *.py[cod] *.orig
|
||||
|
||||
# Include the README
|
||||
# Include the README and CHANGELOG
|
||||
include *.md
|
||||
|
||||
# Include the license file
|
||||
@@ -16,9 +15,7 @@ exclude *.svg
|
||||
recursive-include pytorch_lightning *.py
|
||||
|
||||
# include examples
|
||||
recursive-include pl_examples *.py
|
||||
recursive-include pl_examples *.md
|
||||
recursive-include pl_examples *.sh
|
||||
recursive-include pl_examples *.py *.md *.sh *.txt
|
||||
|
||||
# exclude tests from package
|
||||
recursive-exclude tests *
|
||||
@@ -28,9 +25,12 @@ exclude tests
|
||||
# Exclude the documentation files
|
||||
recursive-exclude docs *
|
||||
exclude docs
|
||||
recursive-include docs/source/_images/logos/ *
|
||||
recursive-include docs/source/_images/general/ pl_overview* tf_* tutorial_*
|
||||
|
||||
# Include the Requirements
|
||||
include requirements.txt
|
||||
include requirements-extra.txt
|
||||
|
||||
# Exclude build configs
|
||||
exclude *.yml
|
||||
@@ -41,3 +41,6 @@ prune .circleci
|
||||
prune notebook*
|
||||
prune temp*
|
||||
prune test*
|
||||
prune benchmark*
|
||||
prune docker
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
<div align="center">
|
||||
|
||||

|
||||

|
||||
|
||||
# PyTorch Lightning
|
||||
|
||||
@@ -9,198 +9,237 @@
|
||||
|
||||
[](https://badge.fury.io/py/pytorch-lightning)
|
||||
[](https://pepy.tech/project/pytorch-lightning)
|
||||
[](https://travis-ci.org/williamFalcon/pytorch-lightning)
|
||||
[](https://ci.appveyor.com/project/williamFalcon/pytorch-lightning)
|
||||
[](https://github.com/williamFalcon/pytorch-lightning/tree/master/tests#running-coverage)
|
||||
[](https://www.codefactor.io/repository/github/borda/pytorch-lightning)
|
||||
[](https://codecov.io/gh/PyTorchLightning/pytorch-lightning)
|
||||
[](https://www.codefactor.io/repository/github/pytorchlightning/pytorch-lightning)
|
||||
|
||||
[](https://pytorch-lightning.readthedocs.io/en/latest)
|
||||
[](https://pytorch-lightning.readthedocs.io/en/stable/)
|
||||
[](https://join.slack.com/t/pytorch-lightning/shared_invite/enQtODU5ODIyNTUzODQwLTFkMDg5Mzc1MDBmNjEzMDgxOTVmYTdhYjA1MDdmODUyOTg2OGQ1ZWZkYTQzODhhNzdhZDA3YmNhMDhlMDY4YzQ)
|
||||
[](https://github.com/williamFalcon/pytorch-lightning/blob/master/LICENSE)
|
||||
[](https://shields.io/)
|
||||
[](https://github.com/PytorchLightning/pytorch-lightning/blob/master/LICENSE)
|
||||
[](https://shields.io/)
|
||||
|
||||
<!--
|
||||
<!--
|
||||
removed until codecov badge isn't empy. likely a config error showing nothing on master.
|
||||
[](https://codecov.io/gh/Borda/pytorch-lightning)
|
||||
-->
|
||||
|
||||
</div>
|
||||
|
||||
---
|
||||
## Continuous Integration
|
||||
<center>
|
||||
|
||||
| System / PyTorch ver. | 1.1 (min. reg) | 1.2 | 1.3 | 1.4 | 1.5 (latest) |
|
||||
| :---: | :---: | :---: | :---: | :---: | :---: |
|
||||
| Linux py3.6 [CPU] | [](https://circleci.com/gh/PyTorchLightning/pytorch-lightning) | [](https://circleci.com/gh/PyTorchLightning/pytorch-lightning) | [](https://circleci.com/gh/PyTorchLightning/pytorch-lightning) | [](https://circleci.com/gh/PyTorchLightning/pytorch-lightning) | [](https://circleci.com/gh/PyTorchLightning/pytorch-lightning) |
|
||||
| Linux py3.7 [GPU] | - | - | - | - | [](http://35.192.60.23/PyTorchLightning/pytorch-lightning) |
|
||||
| Linux py3.6 / py3.7 / py3.8 | [](https://github.com/PyTorchLightning/pytorch-lightning/actions?query=workflow%3A%22CI+testing%22) | - | - | - | [](https://github.com/PyTorchLightning/pytorch-lightning/actions?query=workflow%3A%22CI+testing%22) |
|
||||
| OSX py3.6 / py3.7 / py3.8| [](https://github.com/PyTorchLightning/pytorch-lightning/actions?query=workflow%3A%22CI+testing%22) | - | - | - | [](https://github.com/PyTorchLightning/pytorch-lightning/actions?query=workflow%3A%22CI+testing%22) |
|
||||
| Windows py3.6 / py3.7 / py3.8 | [](https://github.com/PyTorchLightning/pytorch-lightning/actions?query=workflow%3A%22CI+testing%22) | - | - | [](https://github.com/PyTorchLightning/pytorch-lightning/actions?query=workflow%3A%22CI+testing%22) | - |
|
||||
|
||||
</center>
|
||||
|
||||
Simple installation from PyPI
|
||||
```bash
|
||||
pip install pytorch-lightning
|
||||
pip install pytorch-lightning
|
||||
```
|
||||
|
||||
## Docs
|
||||
**[View the docs here](https://williamfalcon.github.io/pytorch-lightning/)**
|
||||
## Docs
|
||||
- [master](https://pytorch-lightning.readthedocs.io/en/latest)
|
||||
- [0.7.6](https://pytorch-lightning.readthedocs.io/en/0.7.6/)
|
||||
- [0.7.5](https://pytorch-lightning.readthedocs.io/en/0.7.5/)
|
||||
- [0.7.3](https://pytorch-lightning.readthedocs.io/en/0.7.3/)
|
||||
- [0.7.1](https://pytorch-lightning.readthedocs.io/en/0.7.1/)
|
||||
- [0.6.0](https://pytorch-lightning.readthedocs.io/en/0.6.0/)
|
||||
- [0.5.3.2](https://pytorch-lightning.readthedocs.io/en/0.5.3.2/)
|
||||
|
||||
## Demo
|
||||
[Copy and run this COLAB!](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=HOk9c4_35FKg)
|
||||
## Refactoring your PyTorch code + benefits + full walk-through
|
||||
[](https://www.youtube.com/watch?v=QHww1JH7IDU)
|
||||
|
||||
## What is it?
|
||||
Lightning is a very lightweight wrapper on PyTorch that decouples the science code from the engineering code. It's more of a style-guide than a framework. By refactoring your code, we can automate most of the non-research code.
|
||||
## Demo
|
||||
Here's a minimal example without a validation or test loop.
|
||||
|
||||
To use Lightning, simply refactor your research code into the [LightningModule](https://github.com/williamFalcon/pytorch-lightning#how-do-i-do-use-it) format (the science) and Lightning will automate the rest (the engineering). Lightning guarantees tested, correct, modern best practices for the automated parts.
|
||||
```python
|
||||
# this is just a plain nn.Module with some structure
|
||||
|
||||
- If you are a researcher, Lightning is infinitely flexible, you can modify everything down to the way .backward is called or distributed is set up.
|
||||
- If you are a scientist or production team, lightning is very simple to use with best practice defaults.
|
||||
class LitClassifier(pl.LightningModule):
|
||||
|
||||
## What does lightning control for me?
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.l1 = torch.nn.Linear(28 * 28, 10)
|
||||
|
||||
Everything in Blue!
|
||||
This is how lightning separates the science (red) from the engineering (blue).
|
||||
def forward(self, x):
|
||||
return torch.relu(self.l1(x.view(x.size(0), -1)))
|
||||
|
||||

|
||||
def training_step(self, batch, batch_nb):
|
||||
x, y = batch
|
||||
loss = F.cross_entropy(self(x), y)
|
||||
tensorboard_logs = {'train_loss': loss}
|
||||
return {'loss': loss, 'log': tensorboard_logs}
|
||||
|
||||
def configure_optimizers(self):
|
||||
return torch.optim.Adam(self.parameters(), lr=0.02)
|
||||
|
||||
# train!
|
||||
train_loader = DataLoader(MNIST(os.getcwd(), train=True, download=True, transform=transforms.ToTensor()), batch_size=32)
|
||||
|
||||
model = LitClassifier()
|
||||
trainer = pl.Trainer(gpus=8, precision=16)
|
||||
trainer.fit(model, train_loader)
|
||||
```
|
||||
|
||||
Other examples:
|
||||
[GAN](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=P0bSmCw57aV5)
|
||||
[BERT](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=7uQVI-xv9Ddj)
|
||||
[DQN](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=NWvMLBDySQI5)
|
||||
[MNIST on TPUs](https://colab.research.google.com/drive/1-_LKx4HwAxl5M6xPJmqAAu444LTDQoa3)
|
||||
|
||||
## What is it?
|
||||
[READ THIS QUICK START PAGE](https://pytorch-lightning.readthedocs.io/en/stable/new-project.html)
|
||||
|
||||
Lightning is a way to organize your PyTorch code to decouple the science code from the engineering.
|
||||
It's more of a PyTorch style-guide than a framework.
|
||||
|
||||
In Lightning, you organize your code into 3 distinct categories:
|
||||
|
||||
1. Research code (goes in the LightningModule).
|
||||
2. Engineering code (you delete, and is handled by the Trainer).
|
||||
3. Non-essential research code (logging, etc... this goes in Callbacks).
|
||||
|
||||
Here's an example of how to refactor your research code into a [LightningModule](https://pytorch-lightning.readthedocs.io/en/latest/lightning-module.html).
|
||||
|
||||

|
||||
|
||||
The rest of the code is automated by the [Trainer](https://pytorch-lightning.readthedocs.io/en/latest/trainer.html)!
|
||||

|
||||
|
||||
## Testing Rigour
|
||||
All the automated code by the Trainer is [tested rigorously with every new PR](https://github.com/PyTorchLightning/pytorch-lightning/tree/master/tests).
|
||||
|
||||
In fact, we also train a few models using a vanilla PyTorch loop and compare with the same model trained using the Trainer to make sure we achieve the EXACT same results. [Check out the parity tests here](https://github.com/PyTorchLightning/pytorch-lightning/tree/master/benchmarks).
|
||||
|
||||
Overall, Lightning guarantees rigorously tested, correct, modern best practices for the automated parts.
|
||||
|
||||
## How flexible is it?
|
||||
As you see, you're just organizing your PyTorch code - there's no abstraction.
|
||||
|
||||
And for the stuff that the Trainer abstracts out, you can [override any part](https://pytorch-lightning.readthedocs.io/en/latest/introduction_guide.html#extensibility) you want to do things like implement your own distributed training, 16-bit precision, or even a custom backward pass.
|
||||
|
||||
For example, here you could do your own backward pass
|
||||
|
||||
```python
|
||||
class LitModel(LightningModule):
|
||||
def optimizer_step(self, current_epoch, batch_idx, optimizer, optimizer_idx,
|
||||
second_order_closure=None):
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
```
|
||||
|
||||
For anything else you might need, we have an extensive [callback system](https://pytorch-lightning.readthedocs.io/en/latest/introduction_guide.html#callbacks) you can use to add arbitrary functionality not implemented by our team in the Trainer.
|
||||
|
||||
## Who is Lightning for?
|
||||
- Professional researchers
|
||||
- Ph.D. students
|
||||
- Corporate production teams
|
||||
|
||||
If you're just getting into deep learning, we recommend you learn PyTorch first! Once you've implemented a few models, come back and use all the advanced features of Lightning :)
|
||||
|
||||
## What does lightning control for me?
|
||||
|
||||
Everything in Blue!
|
||||
This is how lightning separates the science (red) from engineering (blue).
|
||||
|
||||

|
||||
|
||||
## How much effort is it to convert?
|
||||
You're probably tired of switching frameworks at this point. But it is a very quick process to refactor into the Lightning format (ie: hours). [Check out this tutorial](https://towardsdatascience.com/how-to-refactor-your-pytorch-code-to-get-these-42-benefits-of-pytorch-lighting-6fdd0dc97538)
|
||||
If your code is not a huge mess you should be able to organize it into a LightningModule in less than 1 hour.
|
||||
If your code IS a mess, then you needed to clean up anyhow ;)
|
||||
|
||||
## Starting a new project?
|
||||
[Use our seed-project aimed at reproducibility!](https://github.com/williamFalcon/pytorch-lightning-conference-seed)
|
||||
[Check out this step-by-step guide](https://towardsdatascience.com/from-pytorch-to-pytorch-lightning-a-gentle-introduction-b371b7caaf09).
|
||||
[Or watch this video](https://www.youtube.com/watch?v=QHww1JH7IDU).
|
||||
|
||||
|
||||
## Starting a new project?
|
||||
[Use our seed-project aimed at reproducibility!](https://github.com/PytorchLightning/pytorch-lightning-conference-seed)
|
||||
|
||||
## Why do I want to use lightning?
|
||||
Every research project starts the same, a model, a training loop, validation loop, etc. As your research advances, you're likely to need distributed training, 16-bit precision, checkpointing, gradient accumulation, etc.
|
||||
Although your research/production project might start simple, once you add things like GPU AND TPU training, 16-bit precision, etc, you end up spending more time engineering than researching. Lightning automates AND rigorously tests those parts for you.
|
||||
|
||||
Lightning sets up all the boilerplate state-of-the-art training for you so you can focus on the research.
|
||||
## Support
|
||||
- [8 core contributors](https://pytorch-lightning.readthedocs.io/en/latest/governance.html) who are all a mix of professional engineers, Research Scientists, Ph.D. students from top AI labs.
|
||||
- 100+ community contributors.
|
||||
|
||||
---
|
||||
|
||||
## README Table of Contents
|
||||
- [How do I use it](https://github.com/williamFalcon/pytorch-lightning#how-do-i-do-use-it)
|
||||
- [What lightning automates](https://github.com/williamFalcon/pytorch-lightning#what-does-lightning-control-for-me)
|
||||
- [Tensorboard integration](https://github.com/williamFalcon/pytorch-lightning#tensorboard)
|
||||
- [Lightning features](https://github.com/williamFalcon/pytorch-lightning#lightning-automates-all-of-the-following-each-is-also-configurable)
|
||||
- [Examples](https://github.com/williamFalcon/pytorch-lightning#examples)
|
||||
- [Tutorials](https://github.com/williamFalcon/pytorch-lightning#tutorials)
|
||||
- [Contributing](https://github.com/williamFalcon/pytorch-lightning/blob/master/.github/CONTRIBUTING.md)
|
||||
- [Bleeding edge install](https://github.com/williamFalcon/pytorch-lightning#bleeding-edge)
|
||||
- [Lightning Design Principles](https://github.com/williamFalcon/pytorch-lightning#lightning-design-principles)
|
||||
- [Asking for help](https://github.com/williamFalcon/pytorch-lightning#asking-for-help)
|
||||
- [FAQ](https://github.com/williamFalcon/pytorch-lightning#faq)
|
||||
Lightning is also part of the [PyTorch ecosystem](https://pytorch.org/ecosystem/) which requires projects to have solid testing, documentation and support.
|
||||
|
||||
---
|
||||
|
||||
## How do I do use it?
|
||||
Think about Lightning as refactoring your research code instead of using a new framework. The research code goes into a [LightningModule](https://williamfalcon.github.io/pytorch-lightning/LightningModule/RequiredTrainerInterface/) which you fit using a Trainer.
|
||||
## README Table of Contents
|
||||
- [How do I use it](https://github.com/PytorchLightning/pytorch-lightning#how-do-i-do-use-it)
|
||||
- [What lightning automates](https://github.com/PytorchLightning/pytorch-lightning#what-does-lightning-control-for-me)
|
||||
- [Tensorboard integration](https://github.com/PytorchLightning/pytorch-lightning#tensorboard)
|
||||
- [Lightning features](https://github.com/PytorchLightning/pytorch-lightning#lightning-automates-all-of-the-following-each-is-also-configurable)
|
||||
- [Examples](https://github.com/PytorchLightning/pytorch-lightning#examples)
|
||||
- [Tutorials](https://github.com/PytorchLightning/pytorch-lightning#tutorials)
|
||||
- [Asking for help](https://github.com/PytorchLightning/pytorch-lightning#asking-for-help)
|
||||
- [Contributing](https://github.com/PytorchLightning/pytorch-lightning/blob/master/.github/CONTRIBUTING.md)
|
||||
- [Bleeding edge install](https://github.com/PytorchLightning/pytorch-lightning#bleeding-edge)
|
||||
- [Lightning Design Principles](https://github.com/PytorchLightning/pytorch-lightning#lightning-design-principles)
|
||||
- [Lightning team](https://github.com/PytorchLightning/pytorch-lightning#lightning-team)
|
||||
- [FAQ](https://github.com/PytorchLightning/pytorch-lightning#faq)
|
||||
|
||||
The LightningModule defines a *system* such as seq-2-seq, GAN, etc... It can ALSO define a simple classifier such as the example below.
|
||||
---
|
||||
|
||||
## Realistic example
|
||||
Here's how you would organize a realistic PyTorch project into Lightning.
|
||||
|
||||

|
||||
|
||||
The LightningModule defines a *system* such as seq-2-seq, GAN, etc...
|
||||
It can ALSO define a simple classifier.
|
||||
|
||||
In summary, you:
|
||||
|
||||
1. Define a [LightningModule](https://pytorch-lightning.rtfd.io/en/latest/lightning-module.html)
|
||||
```python
|
||||
class LitSystem(pl.LightningModule):
|
||||
|
||||
To use lightning do 2 things:
|
||||
1. [Define a LightningModule](https://williamfalcon.github.io/pytorch-lightning/LightningModule/RequiredTrainerInterface/)
|
||||
**WARNING:** This syntax is for version 0.5.0+ where abbreviations were removed.
|
||||
```python
|
||||
import os
|
||||
|
||||
import torch
|
||||
from torch.nn import functional as F
|
||||
from torch.utils.data import DataLoader
|
||||
from torchvision.datasets import MNIST
|
||||
from torchvision import transforms
|
||||
|
||||
import pytorch_lightning as pl
|
||||
|
||||
class CoolSystem(pl.LightningModule):
|
||||
|
||||
def __init__(self):
|
||||
super(CoolSystem, self).__init__()
|
||||
super().__init__()
|
||||
# not the best model...
|
||||
self.l1 = torch.nn.Linear(28 * 28, 10)
|
||||
|
||||
|
||||
def forward(self, x):
|
||||
return torch.relu(self.l1(x.view(x.size(0), -1)))
|
||||
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
# REQUIRED
|
||||
x, y = batch
|
||||
y_hat = self.forward(x)
|
||||
loss = F.cross_entropy(y_hat, y)
|
||||
tensorboard_logs = {'train_loss': loss}
|
||||
return {'loss': loss, 'log': tensorboard_logs}
|
||||
|
||||
def validation_step(self, batch, batch_idx):
|
||||
# OPTIONAL
|
||||
x, y = batch
|
||||
y_hat = self.forward(x)
|
||||
return {'val_loss': F.cross_entropy(y_hat, y)}
|
||||
|
||||
def validation_end(self, outputs):
|
||||
# OPTIONAL
|
||||
avg_loss = torch.stack([x['val_loss'] for x in outputs]).mean()
|
||||
tensorboard_logs = {'val_loss': avg_loss}
|
||||
return {'avg_val_loss': avg_loss, 'log': tensorboard_logs}
|
||||
|
||||
def configure_optimizers(self):
|
||||
# REQUIRED
|
||||
# can return multiple optimizers and learning_rate schedulers
|
||||
# (LBFGS it is automatically supported, no need for closure function)
|
||||
return torch.optim.Adam(self.parameters(), lr=0.02)
|
||||
|
||||
@pl.data_loader
|
||||
def train_dataloader(self):
|
||||
# REQUIRED
|
||||
return DataLoader(MNIST(os.getcwd(), train=True, download=True, transform=transforms.ToTensor()), batch_size=32)
|
||||
|
||||
@pl.data_loader
|
||||
def val_dataloader(self):
|
||||
# OPTIONAL
|
||||
return DataLoader(MNIST(os.getcwd(), train=True, download=True, transform=transforms.ToTensor()), batch_size=32)
|
||||
|
||||
@pl.data_loader
|
||||
def test_dataloader(self):
|
||||
# OPTIONAL
|
||||
return DataLoader(MNIST(os.getcwd(), train=False, download=True, transform=transforms.ToTensor()), batch_size=32)
|
||||
```
|
||||
2. Fit with a [trainer](https://williamfalcon.github.io/pytorch-lightning/Trainer/)
|
||||
```python
|
||||
from pytorch_lightning import Trainer
|
||||
|
||||
model = CoolSystem()
|
||||
|
||||
# most basic trainer, uses good defaults
|
||||
trainer = Trainer()
|
||||
trainer.fit(model)
|
||||
```
|
||||
|
||||
Trainer sets up a tensorboard logger, early stopping and checkpointing by default (you can modify all of them or
|
||||
use something other than tensorboard).
|
||||
|
||||
Here are more advanced examples
|
||||
```python
|
||||
# train on cpu using only 10% of the data (for demo purposes)
|
||||
trainer = Trainer(max_epochs=1, train_percent_check=0.1)
|
||||
|
||||
# train on 4 gpus (lightning chooses GPUs for you)
|
||||
# trainer = Trainer(max_epochs=1, gpus=4, distributed_backend='ddp')
|
||||
|
||||
# train on 4 gpus (you choose GPUs)
|
||||
# trainer = Trainer(max_epochs=1, gpus=[0, 1, 3, 7], distributed_backend='ddp')
|
||||
|
||||
# train on 32 gpus across 4 nodes (make sure to submit appropriate SLURM job)
|
||||
# trainer = Trainer(max_epochs=1, gpus=8, num_gpu_nodes=4, distributed_backend='ddp')
|
||||
|
||||
# train (1 epoch only here for demo)
|
||||
trainer.fit(model)
|
||||
|
||||
# view tensorboard logs
|
||||
logging.info(f'View tensorboard logs by running\ntensorboard --logdir {os.getcwd()}')
|
||||
logging.info('and going to http://localhost:6006 on your browser')
|
||||
...
|
||||
```
|
||||
|
||||
When you're all done you can even run the test set separately.
|
||||
```python
|
||||
trainer.test()
|
||||
```
|
||||
2. Fit it with a [Trainer](https://pytorch-lightning.rtfd.io/en/latest/pytorch_lightning.trainer.html)
|
||||
```python
|
||||
from pytorch_lightning import Trainer
|
||||
|
||||
**Could be as complex as seq-2-seq + attention**
|
||||
model = LitSystem()
|
||||
|
||||
# most basic trainer, uses good defaults
|
||||
trainer = Trainer()
|
||||
trainer.fit(model)
|
||||
```
|
||||
|
||||
[Check out the COLAB demo here](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=HOk9c4_35FKg)
|
||||
|
||||
## What types of research works?
|
||||
Anything! Remember, that this is just organized PyTorch code.
|
||||
The Training step defines the core complexity found in the training loop.
|
||||
|
||||
#### Could be as complex as a seq2seq
|
||||
|
||||
```python
|
||||
# define what happens for training here
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
|
||||
|
||||
# define your own forward and loss calculation
|
||||
hidden_states = self.encoder(x)
|
||||
|
||||
|
||||
# even as complex as a seq-2-seq + attn model
|
||||
# (this is just a toy, non-working example to illustrate)
|
||||
start_token = '<SOS>'
|
||||
@@ -208,191 +247,154 @@ def training_step(self, batch, batch_idx):
|
||||
loss = 0
|
||||
for step in range(max_seq_len):
|
||||
attn_context = self.attention_nn(hidden_states, start_token)
|
||||
pred = self.decoder(start_token, attn_context, last_hidden)
|
||||
pred = self.decoder(start_token, attn_context, last_hidden)
|
||||
last_hidden = pred
|
||||
pred = self.predict_nn(pred)
|
||||
loss += self.loss(last_hidden, y[step])
|
||||
|
||||
|
||||
#toy example as well
|
||||
loss = loss / max_seq_len
|
||||
return {'loss': loss}
|
||||
return {'loss': loss}
|
||||
```
|
||||
|
||||
**Or as basic as CNN image classification**
|
||||
#### Or as basic as CNN image classification
|
||||
|
||||
```python
|
||||
# define what happens for validation here
|
||||
def validation_step(self, batch, batch_idx):
|
||||
def validation_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
|
||||
|
||||
# or as basic as a CNN classification
|
||||
out = self.forward(x)
|
||||
out = self(x)
|
||||
loss = my_loss(out, y)
|
||||
return {'loss': loss}
|
||||
return {'loss': loss}
|
||||
```
|
||||
|
||||
**And you also decide how to collate the output of all validation steps**
|
||||
|
||||
And without changing a single line of code, you could run on CPUs
|
||||
```python
|
||||
def validation_end(self, outputs):
|
||||
"""
|
||||
Called at the end of validation to aggregate outputs
|
||||
:param outputs: list of individual outputs of each validation step
|
||||
:return:
|
||||
"""
|
||||
val_loss_mean = 0
|
||||
val_acc_mean = 0
|
||||
for output in outputs:
|
||||
val_loss_mean += output['val_loss']
|
||||
val_acc_mean += output['val_acc']
|
||||
|
||||
val_loss_mean /= len(outputs)
|
||||
val_acc_mean /= len(outputs)
|
||||
logs = {'val_loss': val_loss_mean.item(), 'val_acc': val_acc_mean.item()}
|
||||
result = {'log': logs}
|
||||
return result
|
||||
trainer = Trainer(max_epochs=1)
|
||||
```
|
||||
|
||||
## Tensorboard
|
||||
Lightning is fully integrated with tensorboard, MLFlow and supports any logging module.
|
||||
|
||||

|
||||
|
||||
Lightning also adds a text column with all the hyperparameters for this experiment.
|
||||
|
||||

|
||||
|
||||
## Lightning automates all of the following ([each is also configurable](https://williamfalcon.github.io/pytorch-lightning/Trainer/)):
|
||||
|
||||
#### Checkpointing
|
||||
|
||||
- [Checkpoint callback](https://williamfalcon.github.io/pytorch-lightning/Trainer/Checkpointing/#model-saving)
|
||||
- [Model saving](https://williamfalcon.github.io/pytorch-lightning/Trainer/Checkpointing/#model-saving)
|
||||
- [Model loading](https://williamfalcon.github.io/pytorch-lightning/LightningModule/methods/#load-from-metrics)
|
||||
- [Restoring training session](https://williamfalcon.github.io/pytorch-lightning/Trainer/Checkpointing/#restoring-training-session)
|
||||
|
||||
#### Computing cluster (SLURM)
|
||||
|
||||
- [Running grid search on a cluster](https://williamfalcon.github.io/pytorch-lightning/Trainer/SLURM%20Managed%20Cluster#running-grid-search-on-a-cluster)
|
||||
- [Walltime auto-resubmit](https://williamfalcon.github.io/pytorch-lightning/Trainer/SLURM%20Managed%20Cluster#walltime-auto-resubmit)
|
||||
|
||||
#### Debugging
|
||||
|
||||
- [Fast dev run](https://williamfalcon.github.io/pytorch-lightning/Trainer/debugging/#fast-dev-run)
|
||||
- [Inspect gradient norms](https://williamfalcon.github.io/pytorch-lightning/Trainer/debugging/#inspect-gradient-norms)
|
||||
- [Log GPU usage](https://williamfalcon.github.io/pytorch-lightning/Trainer/debugging/#Log-gpu-usage)
|
||||
- [Make model overfit on subset of data](https://williamfalcon.github.io/pytorch-lightning/Trainer/debugging/#make-model-overfit-on-subset-of-data)
|
||||
- [Print the parameter count by layer](https://williamfalcon.github.io/pytorch-lightning/Trainer/debugging/#print-the-parameter-count-by-layer)
|
||||
- [Print which gradients are nan](https://williamfalcon.github.io/pytorch-lightning/Trainer/debugging/#print-which-gradients-are-nan)
|
||||
- [Print input and output size of every module in system](https://williamfalcon.github.io/pytorch-lightning/LightningModule/properties/#example_input_array)
|
||||
|
||||
|
||||
#### Distributed training
|
||||
Or GPUs
|
||||
```python
|
||||
# 8 GPUs
|
||||
trainer = Trainer(max_epochs=1, gpus=8)
|
||||
|
||||
- [Implement Your Own Distributed (DDP) training](https://williamfalcon.github.io/pytorch-lightning/Trainer/hooks/#init_ddp_connection)
|
||||
- [16-bit mixed precision](https://williamfalcon.github.io/pytorch-lightning/Trainer/Distributed%20training/#16-bit-mixed-precision)
|
||||
- [Multi-GPU](https://williamfalcon.github.io/pytorch-lightning/Trainer/Distributed%20training/#Multi-GPU)
|
||||
- [Multi-node](https://williamfalcon.github.io/pytorch-lightning/Trainer/Distributed%20training/#Multi-node)
|
||||
- [Single GPU](https://williamfalcon.github.io/pytorch-lightning/Trainer/Distributed%20training/#single-gpu)
|
||||
- [Self-balancing architecture](https://williamfalcon.github.io/pytorch-lightning/Trainer/Distributed%20training/#self-balancing-architecture)
|
||||
# 256 GPUs
|
||||
trainer = Trainer(max_epochs=1, gpus=8, num_nodes=32)
|
||||
```
|
||||
|
||||
Or TPUs
|
||||
```python
|
||||
trainer = Trainer(num_tpu_cores=8)
|
||||
```
|
||||
|
||||
When you're done training, run the test accuracy
|
||||
```python
|
||||
trainer.test()
|
||||
```
|
||||
|
||||
## Visualization
|
||||
Lightning has out-of-the-box integration with the popular logging/visualizing frameworks
|
||||
|
||||
- [Tensorboard](https://pytorch.org/docs/stable/tensorboard.html)
|
||||
- [MLFlow](https://mlflow.org/)
|
||||
- [Neptune.ai](https://neptune.ai/)
|
||||
- [Comet.ml](https://www.comet.ml/site/)
|
||||
- [Wandb](https://www.wandb.com/)
|
||||
- [Trains](https://github.com/allegroai/trains)
|
||||
- ...
|
||||
|
||||

|
||||
|
||||
|
||||
#### Experiment Logging
|
||||
## Lightning automates 40+ parts of DL/ML research
|
||||
- GPU training
|
||||
- Distributed GPU (cluster) training
|
||||
- TPU training
|
||||
- EarlyStopping
|
||||
- Logging/Visualizing
|
||||
- Checkpointing
|
||||
- Experiment management
|
||||
- [Full list here](https://pytorch-lightning.readthedocs.io/en/latest/#common-use-cases)
|
||||
|
||||
- [Display metrics in progress bar](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#display-metrics-in-progress-bar)
|
||||
- [Log metric row every k batches](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#log-metric-row-every-k-batches)
|
||||
- [Process position](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#process-position)
|
||||
- [Tensorboard support](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#tensorboard-support)
|
||||
- [Save a snapshot of all hyperparameters](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#save-a-snapshot-of-all-hyperparameters)
|
||||
- [Snapshot code for a training run](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#snapshot-code-for-a-training-run)
|
||||
- [Write logs file to csv every k batches](https://williamfalcon.github.io/pytorch-lightning/Trainer/Logging/#write-logs-file-to-csv-every-k-batches)
|
||||
|
||||
#### Training loop
|
||||
## Examples
|
||||
Check out this awesome list of research papers and implementations done with Lightning.
|
||||
|
||||
- [Accumulate gradients](https://williamfalcon.github.io/pytorch-lightning/Trainer/Training%20Loop/#accumulated-gradients)
|
||||
- [Force training for min or max epochs](https://williamfalcon.github.io/pytorch-lightning/Trainer/Training%20Loop/#force-training-for-min-or-max-epochs)
|
||||
- [Early stopping callback](https://williamfalcon.github.io/pytorch-lightning/Trainer/Training%20Loop/#early-stopping)
|
||||
- [Force disable early stop](https://williamfalcon.github.io/pytorch-lightning/Trainer/Training%20Loop/#force-disable-early-stop)
|
||||
- [Gradient Clipping](https://williamfalcon.github.io/pytorch-lightning/Trainer/Training%20Loop/#gradient-clipping)
|
||||
- [Hooks](https://williamfalcon.github.io/pytorch-lightning/Trainer/hooks/)
|
||||
- [Learning rate scheduling](https://williamfalcon.github.io/pytorch-lightning/LightningModule/RequiredTrainerInterface/#configure_optimizers)
|
||||
- [Use multiple optimizers (like GANs)](https://williamfalcon.github.io/pytorch-lightning/LightningModule/RequiredTrainerInterface/#configure_optimizers)
|
||||
- [Set how much of the training set to check (1-100%)](https://williamfalcon.github.io/pytorch-lightning/Trainer/Training%20Loop/#set-how-much-of-the-training-set-to-check)
|
||||
- [Step optimizers at arbitrary intervals](https://williamfalcon.github.io/pytorch-lightning/Trainer/hooks/#optimizer_step)
|
||||
- [Contextual Emotion Detection (DoubleDistilBert)](https://github.com/PyTorchLightning/emotion_transformer)
|
||||
- [Generative Adversarial Network](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=TyYOdg8g77P0)
|
||||
- [Hyperparameter optimization with Optuna](https://github.com/optuna/optuna/blob/master/examples/pytorch_lightning_simple.py)
|
||||
- [Image Inpainting using Partial Convolutions](https://github.com/ryanwongsa/Image-Inpainting)
|
||||
- [MNIST on TPU](https://colab.research.google.com/drive/1-_LKx4HwAxl5M6xPJmqAAu444LTDQoa3#scrollTo=BHBz1_AnamN_)
|
||||
- [NER (transformers, TPU, huggingface)](https://colab.research.google.com/drive/1dBN-wwYUngLYVt985wGs_OKPlK_ANB9D)
|
||||
- [NeuralTexture (CVPR)](https://github.com/PyTorchLightning/neuraltexture)
|
||||
- [Recurrent Attentive Neural Process](https://github.com/PyTorchLightning/attentive-neural-processes)
|
||||
- [Siamese Nets for One-shot Image Recognition](https://github.com/PyTorchLightning/Siamese-Neural-Networks)
|
||||
- [Speech Transformers](https://github.com/PyTorchLightning/speech-transformer-pytorch_lightning)
|
||||
- [Transformers transfer learning (Huggingface)](https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=yr7eaxkF-djf)
|
||||
- [Transformers text classification](https://github.com/ricardorei/lightning-text-classification)
|
||||
- [VAE Library of over 18+ VAE flavors](https://github.com/AntixK/PyTorch-VAE)
|
||||
|
||||
#### Validation loop
|
||||
|
||||
- [Check validation every n epochs](https://williamfalcon.github.io/pytorch-lightning/Trainer/Validation%20loop/#check-validation-every-n-epochs)
|
||||
- [Hooks](https://williamfalcon.github.io/pytorch-lightning/Trainer/hooks/)
|
||||
- [Set how much of the validation set to check](https://williamfalcon.github.io/pytorch-lightning/Trainer/Validation%20loop/#set-how-much-of-the-validation-set-to-check)
|
||||
- [Set how much of the test set to check](https://williamfalcon.github.io/pytorch-lightning/Trainer/Validation%20loop/#set-how-much-of-the-test-set-to-check)
|
||||
- [Set validation check frequency within 1 training epoch](https://williamfalcon.github.io/pytorch-lightning/Trainer/Validation%20loop/#set-validation-check-frequency-within-1-training-epoch)
|
||||
- [Set the number of validation sanity steps](https://williamfalcon.github.io/pytorch-lightning/Trainer/Validation%20loop/#set-the-number-of-validation-sanity-steps)
|
||||
|
||||
#### Testing loop
|
||||
- [Run test set](https://williamfalcon.github.io/pytorch-lightning/Trainer/Testing%20loop/)
|
||||
|
||||
## Examples
|
||||
- [GAN](https://github.com/williamFalcon/pytorch-lightning/tree/master/pl_examples/domain_templates/gan.py)
|
||||
- [MNIST](https://github.com/williamFalcon/pytorch-lightning/tree/master/pl_examples/basic_examples)
|
||||
- [Other projects using Lightning](https://github.com/williamFalcon/pytorch-lightning/network/dependents?package_id=UGFja2FnZS0zNzE3NDU4OTM%3D)
|
||||
- [Multi-node](https://github.com/williamFalcon/pytorch-lightning/tree/master/pl_examples/multi_node_examples)
|
||||
|
||||
## Tutorials
|
||||
- [Basic Lightning use](https://towardsdatascience.com/supercharge-your-ai-research-with-pytorch-lightning-337948a99eec)
|
||||
- [9 key speed features in Pytorch-Lightning](https://towardsdatascience.com/9-tips-for-training-lightning-fast-neural-networks-in-pytorch-8e63a502f565)
|
||||
- [SLURM, multi-node training with Lightning](https://towardsdatascience.com/trivial-multi-node-training-with-pytorch-lightning-ff75dfb809bd)
|
||||
## Tutorials
|
||||
Check out our [introduction guide](https://pytorch-lightning.readthedocs.io/en/latest/introduction_guide.html) to get started.
|
||||
Or jump straight into [our tutorials](https://pytorch-lightning.readthedocs.io/en/latest/#tutorials).
|
||||
|
||||
---
|
||||
|
||||
## Asking for help
|
||||
Welcome to the Lightning community!
|
||||
## Asking for help
|
||||
Welcome to the Lightning community!
|
||||
|
||||
If you have any questions, feel free to:
|
||||
1. [read the docs](https://williamfalcon.github.io/pytorch-lightning/).
|
||||
2. [Search through the issues](https://github.com/williamFalcon/pytorch-lightning/issues?utf8=%E2%9C%93&q=my++question).
|
||||
3. [Ask on stackoverflow](https://stackoverflow.com/questions/ask?guided=false) with the tag pytorch-lightning.
|
||||
If you have any questions, feel free to:
|
||||
1. [read the docs](https://pytorch-lightning.rtfd.io/en/latest/).
|
||||
2. [Search through the issues](https://github.com/PytorchLightning/pytorch-lightning/issues?utf8=%E2%9C%93&q=my++question).
|
||||
3. [Ask on stackoverflow](https://stackoverflow.com/questions/ask?guided=false) with the tag pytorch-lightning.
|
||||
4. [Join our slack](https://join.slack.com/t/pytorch-lightning/shared_invite/enQtODU5ODIyNTUzODQwLTFkMDg5Mzc1MDBmNjEzMDgxOTVmYTdhYjA1MDdmODUyOTg2OGQ1ZWZkYTQzODhhNzdhZDA3YmNhMDhlMDY4YzQ).
|
||||
|
||||
If no one replies to you quickly enough, feel free to post the stackoverflow link to our Gitter chat!
|
||||
---
|
||||
## FAQ
|
||||
**How do I use Lightning for rapid research?**
|
||||
[Here's a walk-through](https://pytorch-lightning.readthedocs.io/en/latest/introduction_guide.html)
|
||||
|
||||
To chat with the rest of us visit our [gitter channel](https://gitter.im/PyTorch-Lightning/community)!
|
||||
|
||||
---
|
||||
## FAQ
|
||||
**How do I use Lightning for rapid research?**
|
||||
[Here's a walk-through](https://williamfalcon.github.io/pytorch-lightning/)
|
||||
|
||||
**Why was Lightning created?**
|
||||
**Why was Lightning created?**
|
||||
Lightning has 3 goals in mind:
|
||||
1. Maximal flexibility while abstracting out the common boilerplate across research projects.
|
||||
2. Reproducibility. If all projects use the LightningModule template, it will be much much easier to understand what's going on and where to look! It will also mean every implementation follows a standard format.
|
||||
3. Democratizing PyTorch power user features. Distributed training? 16-bit? know you need them but don't want to take the time to implement? All good... these come built into Lightning.
|
||||
|
||||
**How does Lightning compare with Ignite and fast.ai?**
|
||||
[Here's a thorough comparison](https://medium.com/@_willfalcon/pytorch-lightning-vs-pytorch-ignite-vs-fast-ai-61dc7480ad8a).
|
||||
1. Maximal flexibility while abstracting out the common boilerplate across research projects.
|
||||
2. Reproducibility. If all projects use the LightningModule template, it will be much much easier to understand what's going on and where to look! It will also mean every implementation follows a standard format.
|
||||
3. Democratizing PyTorch power-user features. Distributed training? 16-bit? know you need them but don't want to take the time to implement? All good... these come built into Lightning.
|
||||
|
||||
**How does Lightning compare with Ignite and fast.ai?**
|
||||
[Here's a thorough comparison](https://medium.com/@_willfalcon/pytorch-lightning-vs-pytorch-ignite-vs-fast-ai-61dc7480ad8a).
|
||||
|
||||
**Is this another library I have to learn?**
|
||||
Nope! We use pure Pytorch everywhere and don't add unecessary abstractions!
|
||||
Nope! We use pure Pytorch everywhere and don't add unnecessary abstractions!
|
||||
|
||||
**Are there plans to support Python 2?**
|
||||
Nope.
|
||||
**Are there plans to support Python 2?**
|
||||
Nope.
|
||||
|
||||
**Are there plans to support virtualenv?**
|
||||
Nope. Please use anaconda or miniconda.
|
||||
```bash
|
||||
conda activate my_env
|
||||
pip install pytorch-lightning
|
||||
```
|
||||
|
||||
**Which PyTorch versions do you support?**
|
||||
- **PyTorch 1.1.0**
|
||||
```bash
|
||||
# install pytorch 1.1.0 using the official instructions
|
||||
|
||||
# install test-tube 0.6.7.6 which supports 1.1.0
|
||||
pip install test-tube==0.6.7.6
|
||||
|
||||
# install latest Lightning version without upgrading deps
|
||||
- **PyTorch 1.1.0**
|
||||
```bash
|
||||
# install pytorch 1.1.0 using the official instructions
|
||||
|
||||
# install test-tube 0.6.7.6 which supports 1.1.0
|
||||
pip install test-tube==0.6.7.6
|
||||
|
||||
# install latest Lightning version without upgrading deps
|
||||
pip install -U --no-deps pytorch-lightning
|
||||
```
|
||||
- **PyTorch 1.2.0, 1.3.0,**
|
||||
Install via pip as normal
|
||||
```
|
||||
- **PyTorch 1.2.0+**
|
||||
```python
|
||||
pip install pytorch-lightning
|
||||
```
|
||||
|
||||
## Custom installation
|
||||
|
||||
@@ -401,29 +403,55 @@ Nope. Please use anaconda or miniconda.
|
||||
If you can't wait for the next release, install the most up to date code with:
|
||||
* using GIT (locally clone whole repo with full history)
|
||||
```bash
|
||||
pip install git+https://github.com/williamFalcon/pytorch-lightning.git@master --upgrade
|
||||
pip install git+https://github.com/PytorchLightning/pytorch-lightning.git@master --upgrade
|
||||
```
|
||||
* using instant zip (last state of the repo without git history)
|
||||
```bash
|
||||
pip install https://github.com/williamFalcon/pytorch-lightning/archive/master.zip --upgrade
|
||||
pip install https://github.com/PytorchLightning/pytorch-lightning/archive/master.zip --upgrade
|
||||
```
|
||||
|
||||
### Any release installation
|
||||
|
||||
You can also install any past release from this repository:
|
||||
You can also install any past release `0.X.Y` from this repository:
|
||||
```bash
|
||||
pip install https://github.com/williamFalcon/pytorch-lightning/archive/0.4.4.zip --upgrade
|
||||
pip install https://github.com/PytorchLightning/pytorch-lightning/archive/0.X.Y.zip --upgrade
|
||||
```
|
||||
|
||||
### Lightning team
|
||||
|
||||
#### Leads
|
||||
- William Falcon [(williamFalcon)](https://github.com/williamFalcon) (Lightning founder)
|
||||
- Jirka Borovec [(Borda)](https://github.com/Borda) (ghost :)
|
||||
- Ethan Harris [(ethanwharris)](https://github.com/ethanwharris) (Torchbearer founder)
|
||||
- Matthew Painter [(MattPainter01)](https://github.com/MattPainter01) (Torchbearer founder)
|
||||
- Justus Schock [(justusschock)](https://github.com/justusschock) (Former Core Member PyTorch Ignite)
|
||||
|
||||
#### Core Maintainers
|
||||
|
||||
- Nick Eggert [(neggert)](https://github.com/neggert)
|
||||
- Jeff Ling [(jeffling)](https://github.com/jeffling)
|
||||
- Jeremy Jordan [(jeremyjordan)](https://github.com/jeremyjordan)
|
||||
- Tullie Murrell [(tullie)](https://github.com/tullie)
|
||||
- Adrian Wälchli [(awaelchli)](https://github.com/awaelchli)
|
||||
|
||||
#### Funding
|
||||
Building open-source software with only a few part-time people is hard! We've secured funding to make sure we can
|
||||
hire a full-time staff, attend conferences, and move faster through implementing features you request.
|
||||
|
||||
Our goal is to build an incredible research platform and a big supportive community. Many open-source projects
|
||||
have gone on to fund operations through things like support and special help for big corporations!
|
||||
|
||||
If you are one of these corporations, please feel free to reach out to will@pytorchlightning.ai!
|
||||
|
||||
## Bibtex
|
||||
If you want to cite the framework feel free to use this (but only if you loved it 😊):
|
||||
```
|
||||
@misc{Falcon2019,
|
||||
author = {Falcon, W.A.},
|
||||
title = {PyTorch Lightning},
|
||||
year = {2019},
|
||||
publisher = {GitHub},
|
||||
journal = {GitHub repository},
|
||||
howpublished = {\url{https://github.com/williamFalcon/pytorch-lightning}}
|
||||
|
||||
```bibtex
|
||||
@article{falcon2019pytorch,
|
||||
title={PyTorch Lightning},
|
||||
author={Falcon, WA},
|
||||
journal={GitHub. Note: https://github. com/williamFalcon/pytorch-lightning Cited by},
|
||||
volume={3},
|
||||
year={2019}
|
||||
}
|
||||
```
|
||||
|
||||
@@ -1,68 +0,0 @@
|
||||
# https://www.appveyor.com/docs/appveyor-yml/
|
||||
environment:
|
||||
|
||||
# SDK v7.0 MSVC Express 2008's SetEnv.cmd script will fail if the
|
||||
# /E:ON and /V:ON options are not enabled in the batch script interpreter
|
||||
# See: http://stackoverflow.com/a/13751649/163740
|
||||
CMD_IN_ENV: "cmd /E:ON /V:ON /C obvci_appveyor_python_build_env.cmd"
|
||||
|
||||
matrix:
|
||||
# Pre-installed Python versions, which Appveyor may upgrade to
|
||||
# a later point release.
|
||||
# See: http://www.appveyor.com/docs/installed-software#python
|
||||
|
||||
|
||||
# - PYTHON: "C:\\Python35-x64"
|
||||
# PYTHON_VERSION: "3.5.x"
|
||||
# PYTHON_ARCH: "64"
|
||||
# TOXENV: "py35"
|
||||
|
||||
- PYTHON: "C:\\Python36-x64"
|
||||
PYTHON_VERSION: "3.6.x"
|
||||
PYTHON_ARCH: "64"
|
||||
TOXENV: "py36"
|
||||
PIP_PYVER: "36"
|
||||
|
||||
- PYTHON: "C:\\Python37-x64"
|
||||
PYTHON_VERSION: "3.7.x"
|
||||
PYTHON_ARCH: "64"
|
||||
TOXENV: "py37"
|
||||
PIP_PYVER: "37"
|
||||
|
||||
build: off
|
||||
|
||||
# https://www.appveyor.com/docs/build-cache/
|
||||
cache:
|
||||
- C:\ProgramData\chocolatey\bin -> appveyor.yml
|
||||
- C:\ProgramData\chocolatey\lib -> appveyor.yml
|
||||
- '%LOCALAPPDATA%\pip\Cache -> appveyor.yml'
|
||||
|
||||
# scripts that run after cloning repository
|
||||
install:
|
||||
# If there is a newer build queued for the same PR, cancel this one.
|
||||
# The AppVeyor 'rollout builds' option is supposed to serve the same
|
||||
# purpose but it is problematic because it tends to cancel builds pushed
|
||||
# directly to master instead of just PR builds (or the converse).
|
||||
- SET PATH=%PYTHON%;%PYTHON%\\Scripts;%path%
|
||||
#- pip install -U --user "pip<19.3"
|
||||
- python -m pip install -r requirements.txt -f https://download.pytorch.org/whl/torch_stable.html
|
||||
- python -m pip install -r ./tests/requirements.txt
|
||||
- python -m pip install pytest-flake8
|
||||
|
||||
# scripts to run before tests (working directory and environment changes
|
||||
# are persisted from the previous steps such as "before_build")
|
||||
before_test:
|
||||
- python --version
|
||||
- pip --version
|
||||
- pip list
|
||||
- dir
|
||||
|
||||
# to run your custom scripts instead of automatic tests
|
||||
test_script:
|
||||
- coverage run --source pytorch_lightning -m py.test pytorch_lightning tests pl_examples -v --doctest-modules --flake8
|
||||
#- python setup.py sdist
|
||||
#- twine check dist/*
|
||||
|
||||
on_success:
|
||||
- coverage report
|
||||
# - codecov
|
||||
@@ -0,0 +1,153 @@
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
from torch.utils.data import Dataset, DataLoader
|
||||
import tests.base.utils as tutils
|
||||
|
||||
from pytorch_lightning import Trainer, LightningModule, seed_everything
|
||||
|
||||
|
||||
class AverageDataset(Dataset):
|
||||
def __init__(self, dataset_len=300, sequence_len=100):
|
||||
self.dataset_len = dataset_len
|
||||
self.sequence_len = sequence_len
|
||||
self.input_seq = torch.randn(dataset_len, sequence_len, 10)
|
||||
top, bottom = self.input_seq.chunk(2, -1)
|
||||
self.output_seq = top + bottom.roll(shifts=1, dims=-1)
|
||||
|
||||
def __len__(self):
|
||||
return self.dataset_len
|
||||
|
||||
def __getitem__(self, item):
|
||||
return self.input_seq[item], self.output_seq[item]
|
||||
|
||||
|
||||
class ParityRNN(LightningModule):
|
||||
def __init__(self):
|
||||
super(ParityRNN, self).__init__()
|
||||
self.rnn = nn.LSTM(10, 20, batch_first=True)
|
||||
self.linear_out = nn.Linear(in_features=20, out_features=5)
|
||||
|
||||
def forward(self, x):
|
||||
seq, last = self.rnn(x)
|
||||
return self.linear_out(seq)
|
||||
|
||||
def training_step(self, batch, batch_nb):
|
||||
x, y = batch
|
||||
y_hat = self(x)
|
||||
loss = F.mse_loss(y_hat, y)
|
||||
return {'loss': loss}
|
||||
|
||||
def configure_optimizers(self):
|
||||
return torch.optim.Adam(self.parameters(), lr=0.02)
|
||||
|
||||
def train_dataloader(self):
|
||||
return DataLoader(AverageDataset(), batch_size=30)
|
||||
|
||||
|
||||
@pytest.mark.skipif(not torch.cuda.is_available(), reason="test requires GPU machine")
|
||||
def test_pytorch_parity(tmpdir):
|
||||
"""
|
||||
Verify that the same pytorch and lightning models achieve the same results
|
||||
:param tmpdir:
|
||||
:return:
|
||||
"""
|
||||
num_epochs = 2
|
||||
num_rums = 3
|
||||
|
||||
lightning_outs, pl_times = lightning_loop(ParityRNN, num_rums, num_epochs)
|
||||
manual_outs, pt_times = vanilla_loop(ParityRNN, num_rums, num_epochs)
|
||||
# make sure the losses match exactly to 5 decimal places
|
||||
for pl_out, pt_out in zip(lightning_outs, manual_outs):
|
||||
np.testing.assert_almost_equal(pl_out, pt_out, 8)
|
||||
|
||||
tutils.assert_speed_parity(pl_times, pt_times, num_epochs)
|
||||
|
||||
|
||||
def vanilla_loop(MODEL, num_runs=10, num_epochs=10):
|
||||
"""
|
||||
Returns an array with the last loss from each epoch for each run
|
||||
"""
|
||||
device = torch.device('cuda' if torch.cuda.is_available() else "cpu")
|
||||
errors = []
|
||||
times = []
|
||||
|
||||
torch.backends.cudnn.deterministic = True
|
||||
for i in range(num_runs):
|
||||
time_start = time.perf_counter()
|
||||
|
||||
# set seed
|
||||
seed = i
|
||||
seed_everything(seed)
|
||||
|
||||
# init model parts
|
||||
model = MODEL()
|
||||
dl = model.train_dataloader()
|
||||
optimizer = model.configure_optimizers()
|
||||
|
||||
# model to GPU
|
||||
model = model.to(device)
|
||||
|
||||
epoch_losses = []
|
||||
for epoch in range(num_epochs):
|
||||
|
||||
# run through full training set
|
||||
for j, batch in enumerate(dl):
|
||||
x, y = batch
|
||||
x = x.cuda(0)
|
||||
y = y.cuda(0)
|
||||
batch = (x, y)
|
||||
|
||||
loss_dict = model.training_step(batch, j)
|
||||
loss = loss_dict['loss']
|
||||
loss.backward()
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
# track last epoch loss
|
||||
epoch_losses.append(loss.item())
|
||||
|
||||
time_end = time.perf_counter()
|
||||
times.append(time_end - time_start)
|
||||
|
||||
errors.append(epoch_losses[-1])
|
||||
|
||||
return errors, times
|
||||
|
||||
|
||||
def lightning_loop(MODEL, num_runs=10, num_epochs=10):
|
||||
errors = []
|
||||
times = []
|
||||
|
||||
for i in range(num_runs):
|
||||
time_start = time.perf_counter()
|
||||
|
||||
# set seed
|
||||
seed = i
|
||||
seed_everything(seed)
|
||||
model = MODEL()
|
||||
|
||||
# init model parts
|
||||
trainer = Trainer(
|
||||
max_epochs=num_epochs,
|
||||
progress_bar_refresh_rate=0,
|
||||
weights_summary=None,
|
||||
gpus=1,
|
||||
early_stop_callback=False,
|
||||
checkpoint_callback=False,
|
||||
distributed_backend='dp',
|
||||
deterministic=True,
|
||||
)
|
||||
trainer.fit(model)
|
||||
|
||||
final_loss = trainer.running_loss.last().item()
|
||||
errors.append(final_loss)
|
||||
|
||||
time_end = time.perf_counter()
|
||||
times.append(time_end - time_start)
|
||||
|
||||
return errors, times
|
||||
@@ -0,0 +1,153 @@
|
||||
import os
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
from torch.utils.data import DataLoader
|
||||
from torchvision import transforms
|
||||
import tests.base.utils as tutils
|
||||
|
||||
from pytorch_lightning import Trainer, LightningModule, seed_everything
|
||||
from tests.base.datasets import TrialMNIST
|
||||
|
||||
|
||||
class ParityMNIST(LightningModule):
|
||||
|
||||
def __init__(self):
|
||||
super(ParityMNIST, self).__init__()
|
||||
self.c_d1 = nn.Linear(in_features=28 * 28, out_features=128)
|
||||
self.c_d1_bn = nn.BatchNorm1d(128)
|
||||
self.c_d1_drop = nn.Dropout(0.3)
|
||||
self.c_d2 = nn.Linear(in_features=128, out_features=10)
|
||||
|
||||
def forward(self, x):
|
||||
x = x.view(x.size(0), -1)
|
||||
x = self.c_d1(x)
|
||||
x = torch.tanh(x)
|
||||
x = self.c_d1_bn(x)
|
||||
x = self.c_d1_drop(x)
|
||||
x = self.c_d2(x)
|
||||
return x
|
||||
|
||||
def training_step(self, batch, batch_nb):
|
||||
x, y = batch
|
||||
y_hat = self(x)
|
||||
loss = F.cross_entropy(y_hat, y)
|
||||
return {'loss': loss}
|
||||
|
||||
def configure_optimizers(self):
|
||||
return torch.optim.Adam(self.parameters(), lr=0.02)
|
||||
|
||||
def train_dataloader(self):
|
||||
return DataLoader(TrialMNIST(train=True,
|
||||
download=True,
|
||||
num_samples=500,
|
||||
digits=list(range(5))),
|
||||
batch_size=128)
|
||||
|
||||
|
||||
@pytest.mark.skipif(not torch.cuda.is_available(), reason="test requires GPU machine")
|
||||
def test_pytorch_parity(tmpdir):
|
||||
"""
|
||||
Verify that the same pytorch and lightning models achieve the same results
|
||||
:param tmpdir:
|
||||
:return:
|
||||
"""
|
||||
num_epochs = 2
|
||||
num_rums = 3
|
||||
lightning_outs, pl_times = lightning_loop(ParityMNIST, num_rums, num_epochs)
|
||||
manual_outs, pt_times = vanilla_loop(ParityMNIST, num_rums, num_epochs)
|
||||
|
||||
# make sure the losses match exactly to 5 decimal places
|
||||
for pl_out, pt_out in zip(lightning_outs, manual_outs):
|
||||
np.testing.assert_almost_equal(pl_out, pt_out, 5)
|
||||
|
||||
# the fist run initialize dataset (download & filter)
|
||||
tutils.assert_speed_parity(pl_times[1:], pt_times[1:], num_epochs)
|
||||
|
||||
|
||||
def vanilla_loop(MODEL, num_runs=10, num_epochs=10):
|
||||
"""
|
||||
Returns an array with the last loss from each epoch for each run
|
||||
"""
|
||||
device = torch.device('cuda' if torch.cuda.is_available() else "cpu")
|
||||
errors = []
|
||||
times = []
|
||||
|
||||
torch.backends.cudnn.deterministic = True
|
||||
for i in range(num_runs):
|
||||
time_start = time.perf_counter()
|
||||
|
||||
# set seed
|
||||
seed = i
|
||||
seed_everything(seed)
|
||||
|
||||
# init model parts
|
||||
model = MODEL()
|
||||
dl = model.train_dataloader()
|
||||
optimizer = model.configure_optimizers()
|
||||
|
||||
# model to GPU
|
||||
model = model.to(device)
|
||||
|
||||
epoch_losses = []
|
||||
for epoch in range(num_epochs):
|
||||
|
||||
# run through full training set
|
||||
for j, batch in enumerate(dl):
|
||||
x, y = batch
|
||||
x = x.cuda(0)
|
||||
y = y.cuda(0)
|
||||
batch = (x, y)
|
||||
|
||||
loss_dict = model.training_step(batch, j)
|
||||
loss = loss_dict['loss']
|
||||
loss.backward()
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
# track last epoch loss
|
||||
epoch_losses.append(loss.item())
|
||||
|
||||
time_end = time.perf_counter()
|
||||
times.append(time_end - time_start)
|
||||
|
||||
errors.append(epoch_losses[-1])
|
||||
|
||||
return errors, times
|
||||
|
||||
|
||||
def lightning_loop(MODEL, num_runs=10, num_epochs=10):
|
||||
errors = []
|
||||
times = []
|
||||
|
||||
for i in range(num_runs):
|
||||
time_start = time.perf_counter()
|
||||
|
||||
# set seed
|
||||
seed = i
|
||||
seed_everything(seed)
|
||||
|
||||
model = MODEL()
|
||||
# init model parts
|
||||
trainer = Trainer(
|
||||
max_epochs=num_epochs,
|
||||
progress_bar_refresh_rate=0,
|
||||
weights_summary=None,
|
||||
gpus=1,
|
||||
early_stop_callback=False,
|
||||
checkpoint_callback=False,
|
||||
deterministic=True,
|
||||
)
|
||||
trainer.fit(model)
|
||||
|
||||
final_loss = trainer.running_loss.last().item()
|
||||
errors.append(final_loss)
|
||||
|
||||
time_end = time.perf_counter()
|
||||
times.append(time_end - time_start)
|
||||
|
||||
return errors, times
|
||||
@@ -0,0 +1,42 @@
|
||||
ARG CUDA_VERSION=10.1
|
||||
FROM nvidia/cuda:${CUDA_VERSION}-base
|
||||
|
||||
# install versions
|
||||
ARG PYTHON_VERSION=3.7
|
||||
ARG PYTORCH_VERSION=1.4
|
||||
ARG LIGHTNING_VERSION=master
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential \
|
||||
cmake \
|
||||
git \
|
||||
curl \
|
||||
ca-certificates
|
||||
|
||||
# add non-root user
|
||||
RUN useradd --create-home --shell /bin/bash containeruser
|
||||
USER containeruser
|
||||
WORKDIR /home/containeruser
|
||||
|
||||
|
||||
# install conda and python
|
||||
RUN curl -o ~/miniconda.sh https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh && \
|
||||
chmod +x ~/miniconda.sh && \
|
||||
~/miniconda.sh -b -p /home/containeruser/conda && \
|
||||
rm ~/miniconda.sh && \
|
||||
/home/containeruser/conda/bin/conda clean -ya && \
|
||||
/home/containeruser/conda/bin/conda install -y python=$PYTHON_VERSION
|
||||
|
||||
# add conda to path
|
||||
ENV PATH /home/containeruser/conda/bin:$PATH
|
||||
|
||||
# install dependencies
|
||||
RUN pip install torch==$PYTORCH_VERSION
|
||||
RUN git clone https://github.com/PyTorchLightning/pytorch-lightning.git --single-branch --branch $LIGHTNING_VERSION && \
|
||||
pip install ./pytorch-lightning && \
|
||||
pip install -r pytorch-lightning/requirements-extra.txt && \
|
||||
rm -rf pytorch-lightning
|
||||
|
||||
RUN python -c "import pytorch_lightning as pl; print(pl.__version__)"
|
||||
|
||||
CMD ["/bin/bash"]
|
||||
@@ -0,0 +1,13 @@
|
||||
## Builds
|
||||
|
||||
You can build it on your own, note it takes lots of time, be prepared.
|
||||
```bash
|
||||
git clone <git-repository>
|
||||
docker image build -t pytorch-lightning:py36 -f docker/Dockerfile --build-arg PYTHON_VERSION=3.6 .
|
||||
```
|
||||
To build other versions, select different Dockerfile.
|
||||
```bash
|
||||
docker image list
|
||||
docker run --rm -it pytorch-lightning:py36 bash
|
||||
docker image rm pytorch-lightning:py36
|
||||
```
|
||||
@@ -0,0 +1 @@
|
||||
make clean ; make html --debug --jobs 2 SPHINXOPTS="-W"
|
||||
@@ -1,9 +1,12 @@
|
||||
sphinx>=1.8.3
|
||||
sphinx>=2.0, <3.0
|
||||
recommonmark # fails with badges
|
||||
m2r # fails with multi-line text
|
||||
nbsphinx
|
||||
pandoc
|
||||
docutils
|
||||
git+https://github.com/Borda/lightning_sphinx_theme.git
|
||||
sphinxcontrib-fulltoc
|
||||
sphinxcontrib-mockautodoc
|
||||
sphinxcontrib-mockautodoc
|
||||
git+https://github.com/PytorchLightning/lightning_sphinx_theme.git
|
||||
# pip_shims
|
||||
sphinx-autodoc-typehints
|
||||
sphinx-paramlinks<0.4.0
|
||||
|
||||
|
After Width: | Height: | Size: 1.1 MiB |
|
Before Width: | Height: | Size: 410 KiB After Width: | Height: | Size: 410 KiB |
|
Before Width: | Height: | Size: 219 KiB After Width: | Height: | Size: 219 KiB |
|
Before Width: | Height: | Size: 214 KiB After Width: | Height: | Size: 214 KiB |
|
After Width: | Height: | Size: 94 KiB |
|
After Width: | Height: | Size: 1.0 MiB |
|
After Width: | Height: | Size: 968 KiB |
@@ -0,0 +1,62 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="no"?>
|
||||
<svg
|
||||
xmlns:dc="http://purl.org/dc/elements/1.1/"
|
||||
xmlns:cc="http://creativecommons.org/ns#"
|
||||
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
||||
xmlns:svg="http://www.w3.org/2000/svg"
|
||||
xmlns="http://www.w3.org/2000/svg"
|
||||
xmlns:sodipodi="http://sodipodi.sourceforge.net/DTD/sodipodi-0.dtd"
|
||||
xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape"
|
||||
id="svg"
|
||||
version="1.1"
|
||||
width="16.000004"
|
||||
height="15.999986"
|
||||
viewBox="0 0 16.000004 15.999986"
|
||||
sodipodi:docname="lightning_icon.svg"
|
||||
inkscape:version="0.92.3 (2405546, 2018-03-11)">
|
||||
<metadata
|
||||
id="metadata13">
|
||||
<rdf:RDF>
|
||||
<cc:Work
|
||||
rdf:about="">
|
||||
<dc:format>image/svg+xml</dc:format>
|
||||
<dc:type
|
||||
rdf:resource="http://purl.org/dc/dcmitype/StillImage" />
|
||||
<dc:title></dc:title>
|
||||
</cc:Work>
|
||||
</rdf:RDF>
|
||||
</metadata>
|
||||
<defs
|
||||
id="defs11" />
|
||||
<sodipodi:namedview
|
||||
pagecolor="#ffffff"
|
||||
bordercolor="#666666"
|
||||
borderopacity="1"
|
||||
objecttolerance="10"
|
||||
gridtolerance="10"
|
||||
guidetolerance="10"
|
||||
inkscape:pageopacity="0"
|
||||
inkscape:pageshadow="2"
|
||||
inkscape:window-width="1920"
|
||||
inkscape:window-height="1028"
|
||||
id="namedview9"
|
||||
showgrid="false"
|
||||
inkscape:zoom="0.59"
|
||||
inkscape:cx="-669.05062"
|
||||
inkscape:cy="373.84245"
|
||||
inkscape:window-x="0"
|
||||
inkscape:window-y="0"
|
||||
inkscape:window-maximized="1"
|
||||
inkscape:current-layer="svg" />
|
||||
<path
|
||||
style="fill:#fbfbfb;fill-rule:evenodd;stroke:none;stroke-width:0.04002798"
|
||||
inkscape:connector-curvature="0"
|
||||
d="m 8.987101,1.723485 c -0.05588,0.03422 -4.121881,4.096544 -4.184645,4.180924 -0.02317,0.0311 -0.04587,0.06016 -0.05044,0.06456 -0.0087,0.0084 -0.07477,0.145063 -0.09679,0.20014 -0.05848,0.146583 -0.05804,0.44387 0.001,0.592413 0.08426,0.21243 0.08826,0.216754 1.576864,1.706274 0.779463,0.779947 1.41719,1.426877 1.41719,1.437604 0,0.0232 -0.253177,0.79848 -0.273873,0.838707 -0.0079,0.0153 -0.01433,0.04087 -0.01433,0.05684 0,0.01597 -0.0059,0.03587 -0.01313,0.04423 -0.0072,0.0084 -0.03678,0.09086 -0.06568,0.18333 -0.02893,0.09246 -0.05904,0.180647 -0.06693,0.195937 -0.0079,0.0153 -0.01437,0.04087 -0.01437,0.05684 0,0.01597 -0.0059,0.03586 -0.01313,0.04423 -0.0072,0.0084 -0.03679,0.09086 -0.06569,0.18333 -0.02893,0.09246 -0.05904,0.180643 -0.06693,0.195937 -0.0079,0.0153 -0.01437,0.04187 -0.01437,0.05908 0,0.0172 -0.0072,0.03574 -0.016,0.04119 -0.0088,0.0054 -0.016,0.02607 -0.016,0.04579 0,0.01973 -0.006,0.04271 -0.0134,0.05108 -0.0074,0.0084 -0.04439,0.112477 -0.08222,0.23136 -0.03787,0.118884 -0.151103,0.461124 -0.251693,0.760534 -0.489984,1.45874 -0.462444,1.36155 -0.413611,1.45938 0.06917,0.138657 0.23128,0.199741 0.358251,0.134974 0.07057,-0.03602 4.143298,-4.099985 4.245368,-4.236242 0.03382,-0.04515 0.09094,-0.165796 0.109916,-0.232123 0.0088,-0.03083 0.0243,-0.08498 0.03442,-0.120363 0.03346,-0.11668 0.0068,-0.361134 -0.0566,-0.520084 C 10.880518,9.229614 10.738898,9.079187 9.372744,7.714673 8.601524,6.944416 7.970523,6.302806 7.970523,6.288916 c 0,-0.01393 0.02817,-0.107833 0.0626,-0.208663 0.03442,-0.100834 0.07881,-0.237367 0.09859,-0.303414 0.0198,-0.06605 0.04207,-0.12693 0.04947,-0.135293 0.0074,-0.0084 0.0135,-0.03133 0.0135,-0.05108 0,-0.01973 0.0072,-0.04035 0.016,-0.04579 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04804 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04804 0.0088,-0.0054 0.016,-0.02707 0.016,-0.04803 0,-0.02097 0.0072,-0.04259 0.016,-0.04803 0.0088,-0.0054 0.016,-0.02397 0.016,-0.04119 0,-0.0172 0.0065,-0.04379 0.0144,-0.05908 0.0079,-0.0153 0.119204,-0.34484 0.247334,-0.73231 C 9.064507,2.979766 9.220177,2.513319 9.28226,2.330632 9.408267,1.960092 9.41367,1.921146 9.35255,1.826839 9.27225,1.703032 9.099973,1.654399 8.986893,1.723566"
|
||||
id="path0" />
|
||||
<path
|
||||
style="fill:#540c8c;fill-rule:evenodd;stroke:none;stroke-width:0.04002798"
|
||||
inkscape:connector-curvature="0"
|
||||
d="m 0.07719102,0.01733399 c -0.02187,0.0111 -0.04875,0.03799 -0.05984,0.05984 -0.0161,0.03173 -0.01937,1.62421701 -0.01633,7.94479601 l 0.0038,7.905086 0.03647,0.03646 0.03646,0.03647 H 8.00241 15.927073 l 0.03646,-0.03647 0.03647,-0.03646 V 8.002393 0.07773399 l -0.03647,-0.03646 -0.03646,-0.03647 -7.905086,-0.0038 c -6.320579,-0.003 -7.91305298,2.4e-4 -7.94479598,0.01633 M 9.193764,1.668208 c 0.259903,0.09046 0.275193,0.212427 0.09363,0.74628 C 8.845834,3.776859 8.388843,5.102846 7.991127,6.302606 L 9.415644,7.72492 c 1.24415,1.242111 1.51682,1.523547 1.51682,1.565414 0,0.0051 0.0133,0.03987 0.02953,0.07718 0.12913,0.296607 0.0877,0.664983 -0.103314,0.91872 -0.141456,0.187933 -4.207341,4.228478 -4.273468,4.246848 -0.139417,0.03871 -0.248653,-0.006 -0.34324,-0.140417 -0.07665,-0.108996 -0.06985,-0.137256 0.287004,-1.194633 0.34663,-1.101761 0.75901,-2.243218 1.08916,-3.290661 0,-0.0078 -0.636164,-0.650377 -1.413707,-1.427921 C 4.877658,7.152643 4.728155,6.995813 4.673718,6.87361 4.661948,6.84718 4.645988,6.81305 4.638168,6.79776 4.630368,6.78246 4.624038,6.75689 4.624038,6.74092 c 0,-0.01597 -0.0076,-0.03659 -0.01687,-0.04587 -0.02253,-0.02253 -0.02253,-0.436904 0,-0.45944 0.0093,-0.0093 0.01687,-0.0327 0.01687,-0.05204 0,-0.0363 0.06917,-0.178363 0.130414,-0.267907 0.07965,-0.1164 4.221831,-4.237681 4.259458,-4.237921 0.02047,-1.2e-4 0.04803,-0.0072 0.06124,-0.01577 0.03147,-0.02033 0.04415,-0.01967 0.118603,0.0062"
|
||||
id="path1"
|
||||
sodipodi:nodetypes="ccscccccccccccscccccscccccccccsssscccc" />
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 6.4 KiB |
@@ -0,0 +1,61 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="no"?>
|
||||
<svg
|
||||
xmlns:dc="http://purl.org/dc/elements/1.1/"
|
||||
xmlns:cc="http://creativecommons.org/ns#"
|
||||
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
||||
xmlns:svg="http://www.w3.org/2000/svg"
|
||||
xmlns="http://www.w3.org/2000/svg"
|
||||
xmlns:sodipodi="http://sodipodi.sourceforge.net/DTD/sodipodi-0.dtd"
|
||||
xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape"
|
||||
id="svg"
|
||||
version="1.1"
|
||||
width="400"
|
||||
height="400"
|
||||
viewBox="0, 0, 400,400"
|
||||
sodipodi:docname="lightning_logo.svg"
|
||||
inkscape:version="0.92.3 (2405546, 2018-03-11)">
|
||||
<metadata
|
||||
id="metadata13">
|
||||
<rdf:RDF>
|
||||
<cc:Work
|
||||
rdf:about="">
|
||||
<dc:format>image/svg+xml</dc:format>
|
||||
<dc:type
|
||||
rdf:resource="http://purl.org/dc/dcmitype/StillImage" />
|
||||
</cc:Work>
|
||||
</rdf:RDF>
|
||||
</metadata>
|
||||
<defs
|
||||
id="defs11" />
|
||||
<sodipodi:namedview
|
||||
pagecolor="#ffffff"
|
||||
bordercolor="#666666"
|
||||
borderopacity="1"
|
||||
objecttolerance="10"
|
||||
gridtolerance="10"
|
||||
guidetolerance="10"
|
||||
inkscape:pageopacity="0"
|
||||
inkscape:pageshadow="2"
|
||||
inkscape:window-width="1920"
|
||||
inkscape:window-height="1028"
|
||||
id="namedview9"
|
||||
showgrid="false"
|
||||
inkscape:zoom="9.44"
|
||||
inkscape:cx="203.07907"
|
||||
inkscape:cy="335.32491"
|
||||
inkscape:window-x="0"
|
||||
inkscape:window-y="0"
|
||||
inkscape:window-maximized="1"
|
||||
inkscape:current-layer="svg" />
|
||||
<path
|
||||
style="fill:#fbfbfb;fill-rule:evenodd;stroke:none"
|
||||
inkscape:connector-curvature="0"
|
||||
d="m 224.6,43.137 c -1.396,0.855 -102.975,102.342 -104.543,104.45 -0.579,0.777 -1.146,1.503 -1.26,1.613 -0.218,0.21 -1.868,3.624 -2.418,5 -1.461,3.662 -1.45,11.089 0.022,14.8 2.105,5.307 2.205,5.415 39.394,42.627 19.473,19.485 35.405,35.647 35.405,35.915 0,0.58 -6.325,19.948 -6.842,20.953 -0.197,0.382 -0.358,1.021 -0.358,1.42 0,0.399 -0.147,0.896 -0.328,1.105 -0.18,0.209 -0.919,2.27 -1.641,4.58 -0.723,2.31 -1.475,4.513 -1.672,4.895 -0.198,0.382 -0.359,1.021 -0.359,1.42 0,0.399 -0.147,0.896 -0.328,1.105 -0.18,0.209 -0.919,2.27 -1.641,4.58 -0.723,2.31 -1.475,4.513 -1.672,4.895 -0.198,0.382 -0.359,1.046 -0.359,1.476 0,0.43 -0.18,0.893 -0.4,1.029 -0.22,0.136 -0.4,0.651 -0.4,1.144 0,0.493 -0.151,1.067 -0.335,1.276 -0.184,0.209 -1.109,2.81 -2.054,5.78 -0.946,2.97 -3.775,11.52 -6.288,19 -12.241,36.443 -11.553,34.015 -10.333,36.459 1.728,3.464 5.778,4.99 8.95,3.372 1.763,-0.9 103.51,-102.428 106.06,-105.832 0.845,-1.128 2.272,-4.142 2.746,-5.799 0.22,-0.77 0.607,-2.123 0.86,-3.007 0.836,-2.915 0.171,-9.022 -1.414,-12.993 -1.493,-3.741 -5.031,-7.499 -39.161,-41.588 C 214.964,173.569 199.2,157.54 199.2,157.193 c 0,-0.348 0.704,-2.694 1.564,-5.213 0.86,-2.519 1.969,-5.93 2.463,-7.58 0.495,-1.65 1.051,-3.171 1.236,-3.38 0.186,-0.209 0.337,-0.783 0.337,-1.276 0,-0.493 0.18,-1.008 0.4,-1.144 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.676 0.4,-1.2 0,-0.524 0.18,-1.064 0.4,-1.2 0.22,-0.136 0.4,-0.599 0.4,-1.029 0,-0.43 0.162,-1.094 0.36,-1.476 0.197,-0.382 2.978,-8.615 6.179,-18.295 3.2,-9.68 7.089,-21.333 8.64,-25.897 3.148,-9.257 3.283,-10.23 1.756,-12.586 -2.006,-3.093 -6.31,-4.308 -9.135,-2.58"
|
||||
id="path0" />
|
||||
<path
|
||||
style="fill:#540c8c;fill-rule:evenodd;stroke:none"
|
||||
inkscape:connector-curvature="0"
|
||||
d="M 2.008,0.513 C 1.462,0.79 0.79,1.462 0.513,2.008 0.111,2.801 0.029,42.585 0.105,200.489 L 0.2,397.978 1.111,398.889 2.022,399.8 H 200 397.978 l 0.911,-0.911 0.911,-0.911 V 200 2.022 L 398.889,1.111 397.978,0.2 200.489,0.105 C 42.585,0.029 2.801,0.111 2.008,0.513 m 227.755,41.243 c 6.493,2.26 6.875,5.307 2.339,18.644 -11.0313,34.035452 -22.44803,67.16196 -32.384,97.135 l 35.588,35.533 c 31.082,31.031 37.894,38.062 37.894,39.108 0,0.128 0.332,0.996 0.738,1.928 3.226,7.41 2.191,16.613 -2.581,22.952 -3.534,4.695 -105.11,105.638 -106.762,106.097 -3.483,0.967 -6.212,-0.15 -8.575,-3.508 -1.915,-2.723 -1.745,-3.429 7.17,-29.845 8.65971,-27.52475 18.96205,-56.04122 27.21,-82.209 0,-0.195 -15.893,-16.248 -35.318,-35.673 -33.146,-33.147 -36.881,-37.065 -38.241,-40.118 -0.294,-0.66 -0.693,-1.513 -0.888,-1.895 -0.194,-0.382 -0.353,-1.021 -0.353,-1.42 0,-0.399 -0.189,-0.914 -0.421,-1.146 -0.563,-0.563 -0.563,-10.915 0,-11.478 0.232,-0.232 0.421,-0.817 0.421,-1.3 0,-0.907 1.728,-4.456 3.258,-6.693 C 120.848,144.96 224.33,42 225.27,41.994 c 0.511,-0.003 1.2,-0.181 1.53,-0.394 0.786,-0.508 1.103,-0.491 2.963,0.156"
|
||||
id="path1"
|
||||
sodipodi:nodetypes="ccscccccccccccscccccscccccccccsssscccc" />
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 5.2 KiB |
|
After Width: | Height: | Size: 16 KiB |
|
Before Width: | Height: | Size: 8.3 KiB After Width: | Height: | Size: 8.3 KiB |
@@ -0,0 +1,62 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="no"?>
|
||||
<svg
|
||||
xmlns:dc="http://purl.org/dc/elements/1.1/"
|
||||
xmlns:cc="http://creativecommons.org/ns#"
|
||||
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
||||
xmlns:svg="http://www.w3.org/2000/svg"
|
||||
xmlns="http://www.w3.org/2000/svg"
|
||||
xmlns:sodipodi="http://sodipodi.sourceforge.net/DTD/sodipodi-0.dtd"
|
||||
xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape"
|
||||
id="svg"
|
||||
version="1.1"
|
||||
width="47.999985"
|
||||
height="47.999943"
|
||||
viewBox="0 0 47.999985 47.999943"
|
||||
sodipodi:docname="lightning_logo.svg"
|
||||
inkscape:version="0.92.3 (2405546, 2018-03-11)">
|
||||
<metadata
|
||||
id="metadata13">
|
||||
<rdf:RDF>
|
||||
<cc:Work
|
||||
rdf:about="">
|
||||
<dc:format>image/svg+xml</dc:format>
|
||||
<dc:type
|
||||
rdf:resource="http://purl.org/dc/dcmitype/StillImage" />
|
||||
<dc:title />
|
||||
</cc:Work>
|
||||
</rdf:RDF>
|
||||
</metadata>
|
||||
<defs
|
||||
id="defs11" />
|
||||
<sodipodi:namedview
|
||||
pagecolor="#ffffff"
|
||||
bordercolor="#666666"
|
||||
borderopacity="1"
|
||||
objecttolerance="10"
|
||||
gridtolerance="10"
|
||||
guidetolerance="10"
|
||||
inkscape:pageopacity="0"
|
||||
inkscape:pageshadow="2"
|
||||
inkscape:window-width="1920"
|
||||
inkscape:window-height="1028"
|
||||
id="namedview9"
|
||||
showgrid="false"
|
||||
inkscape:zoom="0.59"
|
||||
inkscape:cx="-347.96588"
|
||||
inkscape:cy="389.84243"
|
||||
inkscape:window-x="0"
|
||||
inkscape:window-y="0"
|
||||
inkscape:window-maximized="1"
|
||||
inkscape:current-layer="svg" />
|
||||
<path
|
||||
style="fill:#fbfbfb;fill-rule:evenodd;stroke:none;stroke-width:0.12008391"
|
||||
inkscape:connector-curvature="0"
|
||||
d="m 26.961294,5.1704519 c -0.16764,0.10267 -12.36564,12.2896301 -12.55393,12.5427701 -0.0695,0.0933 -0.13762,0.18048 -0.15131,0.19369 -0.0262,0.0252 -0.22432,0.43519 -0.29036,0.60042 -0.17544,0.43975 -0.17412,1.33161 0.003,1.77724 0.25278,0.63729 0.26479,0.65026 4.73059,5.11882 2.33839,2.33984 4.25157,4.28063 4.25157,4.31281 0,0.0696 -0.75953,2.39544 -0.82162,2.51612 -0.0237,0.0459 -0.043,0.12261 -0.043,0.17052 0,0.0479 -0.0177,0.1076 -0.0394,0.13269 -0.0216,0.0251 -0.11035,0.27259 -0.19705,0.54999 -0.0868,0.27739 -0.17713,0.54194 -0.20078,0.58781 -0.0238,0.0459 -0.0431,0.1226 -0.0431,0.17052 0,0.0479 -0.0177,0.10759 -0.0394,0.13269 -0.0216,0.0251 -0.11036,0.27259 -0.19706,0.54999 -0.0868,0.27739 -0.17712,0.54193 -0.20078,0.58781 -0.0238,0.0459 -0.0431,0.1256 -0.0431,0.17724 0,0.0516 -0.0216,0.10723 -0.048,0.12357 -0.0264,0.0163 -0.048,0.0782 -0.048,0.13737 0,0.0592 -0.0181,0.12813 -0.0402,0.15323 -0.0221,0.0251 -0.13318,0.33743 -0.24666,0.69408 -0.1136,0.35665 -0.45331,1.38337 -0.75508,2.2816 -1.46995,4.37622 -1.38733,4.08465 -1.24083,4.37814 0.2075,0.41597 0.69384,0.59922 1.07475,0.40492 0.21171,-0.10807 12.42989,-12.29995 12.7361,-12.70872 0.10147,-0.13545 0.27283,-0.49739 0.32975,-0.69637 0.0264,-0.0925 0.0729,-0.25493 0.10327,-0.36109 0.10039,-0.35004 0.0205,-1.0834 -0.1698,-1.56025 -0.17928,-0.44923 -0.60414,-0.90051 -4.7026,-4.99405 -2.31366,-2.31077 -4.20666,-4.2356 -4.20666,-4.27727 0,-0.0418 0.0845,-0.3235 0.18781,-0.62599 0.10327,-0.3025 0.23644,-0.7121 0.29577,-0.91024 0.0594,-0.19814 0.1262,-0.38079 0.14842,-0.40588 0.0223,-0.0251 0.0405,-0.094 0.0405,-0.15323 0,-0.0592 0.0216,-0.12105 0.048,-0.13738 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.14411 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.14411 0.0264,-0.0163 0.048,-0.0812 0.048,-0.1441 0,-0.0629 0.0216,-0.12777 0.048,-0.1441 0.0264,-0.0163 0.048,-0.0719 0.048,-0.12356 0,-0.0516 0.0195,-0.13137 0.0432,-0.17725 0.0237,-0.0459 0.35761,-1.03452 0.742,-2.19693 0.38427,-1.1624101 0.85128,-2.5617501 1.03753,-3.1098101 0.37802,-1.11162 0.39423,-1.22846 0.21087,-1.51138 -0.24089,-0.37142 -0.75773,-0.51732 -1.09697,-0.30982"
|
||||
id="path0" />
|
||||
<path
|
||||
style="fill:#540c8c;fill-rule:evenodd;stroke:none;stroke-width:0.12008391"
|
||||
inkscape:connector-curvature="0"
|
||||
d="m 0.2315739,0.05200186 c -0.0656,0.0333 -0.14626,0.11396 -0.17952,0.17952 -0.0483,0.0952 -0.0581,4.87265004 -0.049,23.83438014 l 0.0114,23.71525 0.1094,0.10939 0.10939,0.1094 h 23.7739701 23.77398 l 0.10939,-0.1094 0.1094,-0.10939 V 24.007172 0.23320186 l -0.1094,-0.10939 -0.10939,-0.1094 -23.71525,-0.0114 c -18.9617301,-0.009 -23.7391501,7.2e-4 -23.8343801,0.049 M 27.581274,5.0046319 c 0.77971,0.27139 0.82558,0.63728 0.28088,2.23884 -1.32468,4.0871101 -2.69565,8.0650701 -3.8888,11.6643501 l 4.27355,4.26694 c 3.73245,3.72633 4.55046,4.57064 4.55046,4.69624 0,0.0154 0.0399,0.11961 0.0886,0.23153 0.38739,0.88982 0.2631,1.99495 -0.30994,2.75616 -0.42437,0.5638 -12.62202,12.68543 -12.8204,12.74054 -0.41825,0.11613 -0.74596,-0.018 -1.02972,-0.42125 -0.22996,-0.32699 -0.20954,-0.41177 0.86101,-3.5839 1.03989,-3.30528 2.27703,-6.72965 3.26748,-9.87198 0,-0.0234 -1.90849,-1.95113 -4.24112,-4.28376 -3.98031,-3.98042 -4.42882,-4.45091 -4.59213,-4.81752 -0.0353,-0.0793 -0.0832,-0.18169 -0.10664,-0.22756 -0.0233,-0.0459 -0.0424,-0.12261 -0.0424,-0.17052 0,-0.0479 -0.0227,-0.10976 -0.0506,-0.13762 -0.0676,-0.0676 -0.0676,-1.31071 0,-1.37832 0.0279,-0.0279 0.0506,-0.0981 0.0506,-0.15611 0,-0.10891 0.20751,-0.53509 0.39124,-0.80372 0.23896,-0.3492 12.66549,-12.7130401 12.77837,-12.7137601 0.0614,-3.6e-4 0.1441,-0.0217 0.18372,-0.0473 0.0944,-0.061 0.13246,-0.059 0.35581,0.0187"
|
||||
id="path1"
|
||||
sodipodi:nodetypes="ccscccccccccccscccccscccccccccsssscccc" />
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 6.2 KiB |
|
After Width: | Height: | Size: 44 KiB |
|
After Width: | Height: | Size: 66 KiB |
|
After Width: | Height: | Size: 191 KiB |
|
After Width: | Height: | Size: 1.7 MiB |
|
After Width: | Height: | Size: 162 KiB |
|
After Width: | Height: | Size: 116 KiB |
|
After Width: | Height: | Size: 190 KiB |
|
After Width: | Height: | Size: 35 KiB |
|
After Width: | Height: | Size: 17 KiB |
@@ -1,21 +0,0 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<svg xmlns="http://www.w3.org/2000/svg" width="99" height="20">
|
||||
<linearGradient id="b" x2="0" y2="100%">
|
||||
<stop offset="0" stop-color="#bbb" stop-opacity=".1"/>
|
||||
<stop offset="1" stop-opacity=".1"/>
|
||||
</linearGradient>
|
||||
<mask id="a">
|
||||
<rect width="99" height="20" rx="3" fill="#fff"/>
|
||||
</mask>
|
||||
<g mask="url(#a)">
|
||||
<path fill="#555" d="M0 0h63v20H0z"/>
|
||||
<path fill="#4c1" d="M63 0h36v20H63z"/>
|
||||
<path fill="url(#b)" d="M0 0h99v20H0z"/>
|
||||
</g>
|
||||
<g fill="#fff" text-anchor="middle" font-family="DejaVu Sans,Verdana,Geneva,sans-serif" font-size="11">
|
||||
<text x="31.5" y="15" fill="#010101" fill-opacity=".3">coverage</text>
|
||||
<text x="31.5" y="14">coverage</text>
|
||||
<text x="80" y="15" fill="#010101" fill-opacity=".3">99%</text>
|
||||
<text x="80" y="14">99%</text>
|
||||
</g>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 901 B |
|
Before Width: | Height: | Size: 11 KiB |
|
Before Width: | Height: | Size: 2.6 KiB |
|
Before Width: | Height: | Size: 5.4 MiB |
@@ -1,17 +1,18 @@
|
||||
{%- set external_urls = {
|
||||
'github': 'https://github.com/williamFalcon/pytorch-lightning',
|
||||
'github_issues': 'https://github.com/williamFalcon/pytorch-lightning/issues',
|
||||
'contributing': 'https://github.com/williamFalcon/pytorch-lightning/blob/master/CONTRIBUTING.md',
|
||||
'docs': 'https://williamfalcon.github.io/pytorch-lightning',
|
||||
'github': 'https://github.com/PytorchLightning/pytorch-lightning',
|
||||
'github_issues': 'https://github.com/PytorchLightning/pytorch-lightning/issues',
|
||||
'contributing': 'https://github.com/PytorchLightning/pytorch-lightning/blob/master/CONTRIBUTING.md',
|
||||
'governance': 'https://github.com/PytorchLightning/pytorch-lightning/blob/master/governance.md',
|
||||
'docs': 'https://pytorch-lightning.rtfd.io/en/latest',
|
||||
'twitter': 'https://twitter.com/PyTorchLightnin',
|
||||
'discuss': 'https://discuss.pytorch.org',
|
||||
'tutorials': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'previous_pytorch_versions': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'home': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'get_started': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'features': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'blog': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'resources': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'support': 'https://williamfalcon.github.io/pytorch-lightning/',
|
||||
'discuss': 'https://pytorch-lightning.slack.com',
|
||||
'tutorials': 'https://pytorch-lightning.readthedocs.io/en/latest/#tutorials',
|
||||
'previous_pytorch_versions': 'https://pytorch-lightning.rtfd.io/en/latest/',
|
||||
'home': 'https://pytorch-lightning.rtfd.io/en/latest/',
|
||||
'get_started': 'https://pytorch-lightning.readthedocs.io/en/latest/introduction_guide.html',
|
||||
'features': 'https://pytorch-lightning.rtfd.io/en/latest/',
|
||||
'blog': 'https://towardsdatascience.com/@_willfalcon',
|
||||
'resources': 'https://pytorch-lightning.readthedocs.io/en/latest/#community-examples',
|
||||
'support': 'https://pytorch-lightning.rtfd.io/en/latest/',
|
||||
}
|
||||
-%}
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
|
||||
16-bit training
|
||||
=================
|
||||
Lightning offers 16-bit training for CPUs, GPUs and TPUs.
|
||||
|
||||
GPU 16-bit
|
||||
-----------
|
||||
Lightning uses NVIDIA apex to handle 16-bit precision training.
|
||||
|
||||
To use 16-bit precision, do two things:
|
||||
|
||||
1. Install Apex
|
||||
2. Set the "precision" trainer flag.
|
||||
|
||||
Install apex
|
||||
^^^^^^^^^^^^
|
||||
.. code-block:: bash
|
||||
|
||||
$ git clone https://github.com/NVIDIA/apex
|
||||
$ cd apex
|
||||
|
||||
# ------------------------
|
||||
# OPTIONAL: on your cluster you might need to load cuda 10 or 9
|
||||
# depending on how you installed PyTorch
|
||||
|
||||
# see available modules
|
||||
module avail
|
||||
|
||||
# load correct cuda before install
|
||||
module load cuda-10.0
|
||||
# ------------------------
|
||||
|
||||
# make sure you've loaded a cuda version > 4.0 and < 7.0
|
||||
module load gcc-6.1.0
|
||||
|
||||
$ pip install -v --no-cache-dir --global-option="--cpp_ext" --global-option="--cuda_ext" ./
|
||||
|
||||
|
||||
Enable 16-bit
|
||||
^^^^^^^^^^^^^
|
||||
|
||||
.. testcode::
|
||||
|
||||
# turn on 16-bit
|
||||
trainer = Trainer(amp_level='O1', precision=16)
|
||||
|
||||
If you need to configure the apex init for your particular use case or want to use a different way of doing
|
||||
16-bit training, override :meth:`pytorch_lightning.core.LightningModule.configure_apex`.
|
||||
|
||||
TPU 16-bit
|
||||
----------
|
||||
16-bit on TPus is much simpler. To use 16-bit with TPUs set precision to 16 when using the tpu flag
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT
|
||||
trainer = Trainer(num_tpu_cores=8, precision=32)
|
||||
|
||||
# turn on 16-bit
|
||||
trainer = Trainer(num_tpu_cores=8, precision=16)
|
||||
@@ -0,0 +1,101 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.callbacks.base import Callback
|
||||
|
||||
.. role:: hidden
|
||||
:class: hidden-section
|
||||
|
||||
.. _callbacks:
|
||||
|
||||
Callbacks
|
||||
=========
|
||||
|
||||
Lightning has a callback system to execute arbitrary code. Callbacks should capture NON-ESSENTIAL
|
||||
logic that is NOT required for your :class:`~pytorch_lightning.core.LightningModule` to run.
|
||||
|
||||
An overall Lightning system should have:
|
||||
|
||||
1. Trainer for all engineering
|
||||
2. LightningModule for all research code.
|
||||
3. Callbacks for non-essential code.
|
||||
|
||||
|
||||
Example:
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyPrintingCallback(Callback):
|
||||
|
||||
def on_init_start(self, trainer):
|
||||
print('Starting to init trainer!')
|
||||
|
||||
def on_init_end(self, trainer):
|
||||
print('trainer is init now')
|
||||
|
||||
def on_train_end(self, trainer, pl_module):
|
||||
print('do something when training ends')
|
||||
|
||||
trainer = Trainer(callbacks=[MyPrintingCallback()])
|
||||
|
||||
.. testoutput::
|
||||
|
||||
Starting to init trainer!
|
||||
trainer is init now
|
||||
|
||||
We successfully extended functionality without polluting our super clean
|
||||
:class:`~pytorch_lightning.core.LightningModule` research code.
|
||||
|
||||
---------
|
||||
|
||||
.. automodule:: pytorch_lightning.callbacks.base
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_del_model,
|
||||
_save_model,
|
||||
_abc_impl,
|
||||
check_monitor_top_k,
|
||||
|
||||
---------
|
||||
|
||||
.. automodule:: pytorch_lightning.callbacks.early_stopping
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_del_model,
|
||||
_save_model,
|
||||
_abc_impl,
|
||||
check_monitor_top_k,
|
||||
|
||||
---------
|
||||
|
||||
.. automodule:: pytorch_lightning.callbacks.model_checkpoint
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_del_model,
|
||||
_save_model,
|
||||
_abc_impl,
|
||||
check_monitor_top_k,
|
||||
|
||||
---------
|
||||
|
||||
.. automodule:: pytorch_lightning.callbacks.gradient_accumulation_scheduler
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_del_model,
|
||||
_save_model,
|
||||
_abc_impl,
|
||||
check_monitor_top_k,
|
||||
|
||||
---------
|
||||
|
||||
.. automodule:: pytorch_lightning.callbacks.progress
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
|
||||
---------
|
||||
|
||||
.. automodule:: pytorch_lightning.callbacks.lr_logger
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_extract_lr,
|
||||
_find_names
|
||||
@@ -0,0 +1,85 @@
|
||||
.. testsetup:: *
|
||||
|
||||
import torch
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.callbacks.base import Callback
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
|
||||
def train_dataloader():
|
||||
pass
|
||||
|
||||
def val_dataloader():
|
||||
pass
|
||||
|
||||
|
||||
Child Modules
|
||||
-------------
|
||||
Research projects tend to test different approaches to the same dataset.
|
||||
This is very easy to do in Lightning with inheritance.
|
||||
|
||||
For example, imagine we now want to train an Autoencoder to use as a feature extractor for MNIST images.
|
||||
Recall that `LitMNIST` already defines all the dataloading etc... The only things
|
||||
that change in the `Autoencoder` model are the init, forward, training, validation and test step.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class Encoder(torch.nn.Module):
|
||||
pass
|
||||
|
||||
class Decoder(torch.nn.Module):
|
||||
pass
|
||||
|
||||
class AutoEncoder(LitMNIST):
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.encoder = Encoder()
|
||||
self.decoder = Decoder()
|
||||
|
||||
def forward(self, x):
|
||||
generated = self.decoder(x)
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, _ = batch
|
||||
|
||||
representation = self.encoder(x)
|
||||
x_hat = self(representation)
|
||||
|
||||
loss = MSE(x, x_hat)
|
||||
return loss
|
||||
|
||||
def validation_step(self, batch, batch_idx):
|
||||
return self._shared_eval(batch, batch_idx, 'val')
|
||||
|
||||
def test_step(self, batch, batch_idx):
|
||||
return self._shared_eval(batch, batch_idx, 'test')
|
||||
|
||||
def _shared_eval(self, batch, batch_idx, prefix):
|
||||
x, y = batch
|
||||
representation = self.encoder(x)
|
||||
x_hat = self(representation)
|
||||
|
||||
loss = F.nll_loss(logits, y)
|
||||
return {f'{prefix}_loss': loss}
|
||||
|
||||
|
||||
and we can train this using the same trainer
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
autoencoder = AutoEncoder()
|
||||
trainer = Trainer()
|
||||
trainer.fit(autoencoder)
|
||||
|
||||
And remember that the forward method is to define the practical use of a LightningModule.
|
||||
In this case, we want to use the `AutoEncoder` to extract image representations
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
some_images = torch.Tensor(32, 1, 28, 28)
|
||||
representations = autoencoder(some_images)
|
||||
@@ -21,6 +21,7 @@ import inspect
|
||||
# import m2r
|
||||
import builtins
|
||||
import pt_lightning_sphinx_theme
|
||||
from sphinx.ext import apidoc
|
||||
|
||||
PATH_HERE = os.path.abspath(os.path.dirname(__file__))
|
||||
PATH_ROOT = os.path.join(PATH_HERE, '..', '..')
|
||||
@@ -28,6 +29,8 @@ sys.path.insert(0, os.path.abspath(PATH_ROOT))
|
||||
|
||||
builtins.__LIGHTNING_SETUP__ = True
|
||||
|
||||
SPHINX_MOCK_REQUIREMENTS = int(os.environ.get('SPHINX_MOCK_REQUIREMENTS', True))
|
||||
|
||||
import pytorch_lightning # noqa: E402
|
||||
|
||||
# -- Project documents -------------------------------------------------------
|
||||
@@ -62,19 +65,18 @@ version = pytorch_lightning.__version__
|
||||
# The full version, including alpha/beta/rc tags
|
||||
release = pytorch_lightning.__version__
|
||||
|
||||
|
||||
# -- General configuration ---------------------------------------------------
|
||||
|
||||
# If your documentation needs a minimal Sphinx version, state it here.
|
||||
|
||||
needs_sphinx = '1.4'
|
||||
needs_sphinx = '2.0'
|
||||
|
||||
# Add any Sphinx extension module names here, as strings. They can be
|
||||
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||
# ones.
|
||||
extensions = [
|
||||
'sphinx.ext.autodoc',
|
||||
'sphinxcontrib.mockautodoc',
|
||||
# 'sphinxcontrib.mockautodoc', # raises error: directive 'automodule' is already registered ...
|
||||
# 'sphinxcontrib.fulltoc', # breaks pytorch-theme with unexpected kw argument 'titles_only'
|
||||
'sphinx.ext.doctest',
|
||||
'sphinx.ext.intersphinx',
|
||||
@@ -84,8 +86,11 @@ extensions = [
|
||||
'sphinx.ext.autosummary',
|
||||
'sphinx.ext.napoleon',
|
||||
'recommonmark',
|
||||
'sphinx.ext.autosectionlabel',
|
||||
# 'm2r',
|
||||
'nbsphinx',
|
||||
'sphinx_autodoc_typehints',
|
||||
'sphinx_paramlinks',
|
||||
]
|
||||
|
||||
# Add any paths that contain templates here, relative to this directory.
|
||||
@@ -97,6 +102,7 @@ templates_path = ['_templates']
|
||||
# they should be run at build time.
|
||||
nbsphinx_execute = 'never'
|
||||
nbsphinx_allow_errors = True
|
||||
nbsphinx_requirejs_path = ''
|
||||
|
||||
# The suffix(es) of source filenames.
|
||||
# You can specify multiple suffix as a list of string:
|
||||
@@ -123,12 +129,24 @@ language = None
|
||||
# List of patterns, relative to source directory, that match files and
|
||||
# directories to ignore when looking for source files.
|
||||
# This pattern also affects html_static_path and html_extra_path.
|
||||
exclude_patterns = ['*.test_*']
|
||||
exclude_patterns = [
|
||||
'api/pytorch_lightning.rst',
|
||||
'api/pl_examples.*',
|
||||
'api/modules.rst',
|
||||
|
||||
# deprecated/renamed:
|
||||
'api/pytorch_lightning.loggers.comet_logger.rst', # TODO: remove in v0.8.0
|
||||
'api/pytorch_lightning.loggers.mlflow_logger.rst', # TODO: remove in v0.8.0
|
||||
'api/pytorch_lightning.loggers.test_tube_logger.rst', # TODO: remove in v0.8.0
|
||||
'api/pytorch_lightning.callbacks.pt_callbacks.*', # TODO: remove in v0.8.0
|
||||
'api/pytorch_lightning.pt_overrides.*', # TODO: remove in v0.8.0
|
||||
'api/pytorch_lightning.root_module.*', # TODO: remove in v0.8.0
|
||||
'api/pytorch_lightning.logging.*', # TODO: remove in v0.8.0
|
||||
]
|
||||
|
||||
# The name of the Pygments (syntax highlighting) style to use.
|
||||
pygments_style = None
|
||||
|
||||
|
||||
# -- Options for HTML output -------------------------------------------------
|
||||
|
||||
# The theme to use for HTML and HTML Help pages. See the documentation for
|
||||
@@ -151,12 +169,12 @@ html_theme_options = {
|
||||
'logo_only': False,
|
||||
}
|
||||
|
||||
html_logo = '_static/images/lightning_logo_small.png'
|
||||
html_logo = '_images/logos/lightning_logo-name.svg'
|
||||
|
||||
# Add any paths that contain custom static files (such as style sheets) here,
|
||||
# relative to this directory. They are copied after the builtin static files,
|
||||
# so a file named "default.css" will overwrite the builtin "default.css".
|
||||
html_static_path = ['_static']
|
||||
html_static_path = ['_images', '_templates']
|
||||
|
||||
# Custom sidebar templates, must be a dictionary that maps document names
|
||||
# to template names.
|
||||
@@ -174,7 +192,6 @@ html_static_path = ['_static']
|
||||
# Output file base name for HTML help builder.
|
||||
htmlhelp_basename = project + '-doc'
|
||||
|
||||
|
||||
# -- Options for LaTeX output ------------------------------------------------
|
||||
|
||||
latex_elements = {
|
||||
@@ -198,7 +215,6 @@ latex_documents = [
|
||||
(master_doc, project + '.tex', project + ' Documentation', author, 'manual'),
|
||||
]
|
||||
|
||||
|
||||
# -- Options for manual page output ------------------------------------------
|
||||
|
||||
# One entry per manual page. List of tuples
|
||||
@@ -207,7 +223,6 @@ man_pages = [
|
||||
(master_doc, project, project + ' Documentation', [author], 1)
|
||||
]
|
||||
|
||||
|
||||
# -- Options for Texinfo output ----------------------------------------------
|
||||
|
||||
# Grouping the document tree into Texinfo files. List of tuples
|
||||
@@ -218,7 +233,6 @@ texinfo_documents = [
|
||||
'One line description of project.', 'Miscellaneous'),
|
||||
]
|
||||
|
||||
|
||||
# -- Options for Epub output -------------------------------------------------
|
||||
|
||||
# Bibliographic Dublin Core info.
|
||||
@@ -236,13 +250,16 @@ epub_title = project
|
||||
# A list of files that should not be packed into the epub file.
|
||||
epub_exclude_files = ['search.html']
|
||||
|
||||
|
||||
# -- Extension configuration -------------------------------------------------
|
||||
|
||||
# -- Options for intersphinx extension ---------------------------------------
|
||||
|
||||
# Example configuration for intersphinx: refer to the Python standard library.
|
||||
intersphinx_mapping = {'https://docs.python.org/': None}
|
||||
intersphinx_mapping = {
|
||||
'python': ('https://docs.python.org/3', None),
|
||||
'torch': ('https://pytorch.org/docs/stable/', None),
|
||||
'numpy': ('https://docs.scipy.org/doc/numpy/', None),
|
||||
'PIL': ('https://pillow.readthedocs.io/en/stable/', None),
|
||||
}
|
||||
|
||||
# -- Options for todo extension ----------------------------------------------
|
||||
|
||||
@@ -250,32 +267,32 @@ intersphinx_mapping = {'https://docs.python.org/': None}
|
||||
todo_include_todos = True
|
||||
|
||||
|
||||
# https://github.com/rtfd/readthedocs.org/issues/1139
|
||||
# I use sphinx-apidoc to auto-generate API documentation for my project.
|
||||
# Right now I have to commit these auto-generated files to my repository
|
||||
# so that RTD can build them into HTML docs. It'd be cool if RTD could run
|
||||
# sphinx-apidoc for me, since it's easy to forget to regen API docs
|
||||
# and commit them to my repo after making changes to my code.
|
||||
|
||||
# packages for which sphinx-apidoc should generate the docs (.rst files)
|
||||
PACKAGES = [
|
||||
pytorch_lightning.__name__,
|
||||
'pl_examples',
|
||||
]
|
||||
|
||||
apidoc_output_folder = os.path.join(PATH_HERE, 'api')
|
||||
|
||||
|
||||
def run_apidoc(_):
|
||||
sys.path.insert(0, apidoc_output_folder)
|
||||
|
||||
# delete api-doc files before generating them
|
||||
if os.path.exists(apidoc_output_folder):
|
||||
shutil.rmtree(apidoc_output_folder)
|
||||
|
||||
for pkg in PACKAGES:
|
||||
argv = ['-e', '-o', PATH_HERE, os.path.join(PATH_HERE, PATH_ROOT, pkg),
|
||||
'**/test_*', '--force', '--private', '--module-first']
|
||||
try:
|
||||
# Sphinx 1.7+
|
||||
from sphinx.ext import apidoc
|
||||
apidoc.main(argv)
|
||||
except ImportError:
|
||||
# Sphinx 1.6 (and earlier)
|
||||
from sphinx import apidoc
|
||||
argv.insert(0, apidoc.__file__)
|
||||
apidoc.main(argv)
|
||||
argv = ['-e',
|
||||
'-o', apidoc_output_folder,
|
||||
os.path.join(PATH_ROOT, pkg),
|
||||
'**/test_*',
|
||||
'--force',
|
||||
'--private',
|
||||
'--module-first']
|
||||
|
||||
apidoc.main(argv)
|
||||
|
||||
|
||||
def setup(app):
|
||||
@@ -290,27 +307,40 @@ for path_ipynb in glob.glob(os.path.join(PATH_ROOT, 'notebooks', '*.ipynb')):
|
||||
path_ipynb2 = os.path.join(path_nbs, os.path.basename(path_ipynb))
|
||||
shutil.copy(path_ipynb, path_ipynb2)
|
||||
|
||||
|
||||
# Ignoring Third-party packages
|
||||
# https://stackoverflow.com/questions/15889621/sphinx-how-to-exclude-imports-in-automodule
|
||||
def package_list_from_file(file):
|
||||
mocked_packages = []
|
||||
with open(file, 'r') as fp:
|
||||
for ln in fp.readlines():
|
||||
found = [ln.index(ch) for ch in list(',=<>#') if ch in ln]
|
||||
pkg = ln[:min(found)] if found else ln
|
||||
if pkg.rstrip():
|
||||
mocked_packages.append(pkg.rstrip())
|
||||
return mocked_packages
|
||||
|
||||
MOCK_REQUIRE_PACKAGES = []
|
||||
with open(os.path.join(PATH_ROOT, 'requirements.txt'), 'r') as fp:
|
||||
for ln in fp.readlines():
|
||||
found = [ln.index(ch) for ch in list(',=<>#') if ch in ln]
|
||||
pkg = ln[:min(found)] if found else ln
|
||||
if pkg.rstrip():
|
||||
MOCK_REQUIRE_PACKAGES.append(pkg.rstrip())
|
||||
|
||||
# TODO: better parse from package since the import name and package name may differ
|
||||
MOCK_MANUAL_PACKAGES = ['torch', 'torchvision', 'sklearn', 'test_tube', 'mlflow', 'comet_ml']
|
||||
autodoc_mock_imports = MOCK_REQUIRE_PACKAGES + MOCK_MANUAL_PACKAGES
|
||||
# for mod_name in MOCK_REQUIRE_PACKAGES:
|
||||
# sys.modules[mod_name] = mock.Mock()
|
||||
MOCK_PACKAGES = []
|
||||
if SPHINX_MOCK_REQUIREMENTS:
|
||||
# mock also base packages when we are on RTD since we don't install them there
|
||||
MOCK_PACKAGES += package_list_from_file(os.path.join(PATH_ROOT, 'requirements.txt'))
|
||||
MOCK_PACKAGES += package_list_from_file(os.path.join(PATH_ROOT, 'requirements-extra.txt'))
|
||||
|
||||
MOCK_MANUAL_PACKAGES = [
|
||||
'torchvision',
|
||||
'PIL',
|
||||
# packages with different package name compare to import name
|
||||
'yaml',
|
||||
'comet_ml',
|
||||
'neptune',
|
||||
]
|
||||
autodoc_mock_imports = MOCK_PACKAGES + MOCK_MANUAL_PACKAGES
|
||||
|
||||
|
||||
# Options for the linkcode extension
|
||||
# ----------------------------------
|
||||
github_user = 'williamFalcon'
|
||||
github_user = 'PyTorchLightning'
|
||||
github_repo = project
|
||||
|
||||
|
||||
@@ -325,7 +355,7 @@ def linkcode_resolve(domain, info):
|
||||
obj = getattr(obj, part)
|
||||
fname = inspect.getsourcefile(obj)
|
||||
# https://github.com/rtfd/readthedocs.org/issues/5735
|
||||
if any([s in fname for s in ('readthedocs', 'checkouts')]):
|
||||
if any([s in fname for s in ('readthedocs', 'rtfd', 'checkouts')]):
|
||||
# /home/docs/checkouts/readthedocs.org/user_builds/pytorch_lightning/checkouts/
|
||||
# devel/pytorch_lightning/utilities/cls_experiment.py#L26-L176
|
||||
path_top = os.path.abspath(os.path.join('..', '..', '..'))
|
||||
@@ -345,13 +375,51 @@ def linkcode_resolve(domain, info):
|
||||
# import subprocess
|
||||
# tag = subprocess.Popen(['git', 'rev-parse', 'HEAD'], stdout=subprocess.PIPE,
|
||||
# universal_newlines=True).communicate()[0][:-1]
|
||||
branch = filename.split('/')[0]
|
||||
# do mapping from latest tags to master
|
||||
branch = {'latest': 'master', 'stable': 'master'}.get(branch, branch)
|
||||
filename = '/'.join([branch] + filename.split('/')[1:])
|
||||
return "https://github.com/%s/%s/blob/%s" \
|
||||
% (github_user, github_repo, filename)
|
||||
|
||||
|
||||
autodoc_member_order = 'groupwise'
|
||||
autoclass_content = 'both'
|
||||
autodoc_default_flags = [
|
||||
'members', 'undoc-members', 'show-inheritance', 'private-members',
|
||||
# 'special-members', 'inherited-members'
|
||||
]
|
||||
# the options are fixed and will be soon in release,
|
||||
# see https://github.com/sphinx-doc/sphinx/issues/5459
|
||||
autodoc_default_options = {
|
||||
'members': None,
|
||||
'methods': None,
|
||||
# 'attributes': None,
|
||||
'special-members': '__call__',
|
||||
'exclude-members': '_abc_impl',
|
||||
'show-inheritance': True,
|
||||
'private-members': True,
|
||||
'noindex': True,
|
||||
}
|
||||
|
||||
# Sphinx will add “permalinks” for each heading and description environment as paragraph signs that
|
||||
# become visible when the mouse hovers over them.
|
||||
# This value determines the text for the permalink; it defaults to "¶". Set it to None or the empty
|
||||
# string to disable permalinks.
|
||||
# https://www.sphinx-doc.org/en/master/usage/configuration.html#confval-html_add_permalinks
|
||||
html_add_permalinks = "¶"
|
||||
|
||||
# True to prefix each section label with the name of the document it is in, followed by a colon.
|
||||
# For example, index:Introduction for a section called Introduction that appears in document index.rst.
|
||||
# Useful for avoiding ambiguity when the same section heading appears in different documents.
|
||||
# http://www.sphinx-doc.org/en/master/usage/extensions/autosectionlabel.html
|
||||
autosectionlabel_prefix_document = True
|
||||
|
||||
# only run doctests marked with a ".. doctest::" directive
|
||||
doctest_test_doctest_blocks = ''
|
||||
doctest_global_setup = """
|
||||
|
||||
import importlib
|
||||
import os
|
||||
import torch
|
||||
|
||||
TORCHVISION_AVAILABLE = importlib.util.find_spec('torchvision')
|
||||
|
||||
"""
|
||||
coverage_skip_undoc_in_source = True
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
Debugging
|
||||
=========
|
||||
The following are flags that make debugging much easier.
|
||||
|
||||
Fast dev run
|
||||
------------
|
||||
This flag runs a "unit test" by running 1 training batch and 1 validation batch.
|
||||
The point is to detect any bugs in the training/validation loop without having to wait for
|
||||
a full epoch to crash.
|
||||
|
||||
(See: :paramref:`~pytorch_lightning.trainer.trainer.Trainer.fast_dev_run`
|
||||
argument of :class:`~pytorch_lightning.trainer.trainer.Trainer`)
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(fast_dev_run=True)
|
||||
|
||||
Inspect gradient norms
|
||||
----------------------
|
||||
Logs (to a logger), the norm of each weight matrix.
|
||||
|
||||
(See: :paramref:`~pytorch_lightning.trainer.trainer.Trainer.track_grad_norm`
|
||||
argument of :class:`~pytorch_lightning.trainer.trainer.Trainer`)
|
||||
|
||||
.. testcode::
|
||||
|
||||
# the 2-norm
|
||||
trainer = Trainer(track_grad_norm=2)
|
||||
|
||||
Log GPU usage
|
||||
-------------
|
||||
Logs (to a logger) the GPU usage for each GPU on the master machine.
|
||||
|
||||
(See: :paramref:`~pytorch_lightning.trainer.trainer.Trainer.log_gpu_memory`
|
||||
argument of :class:`~pytorch_lightning.trainer.trainer.Trainer`)
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(log_gpu_memory=True)
|
||||
|
||||
Make model overfit on subset of data
|
||||
------------------------------------
|
||||
|
||||
A good debugging technique is to take a tiny portion of your data (say 2 samples per class),
|
||||
and try to get your model to overfit. If it can't, it's a sign it won't work with large datasets.
|
||||
|
||||
(See: :paramref:`~pytorch_lightning.trainer.trainer.Trainer.overfit_pct`
|
||||
argument of :class:`~pytorch_lightning.trainer.trainer.Trainer`)
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(overfit_pct=0.01)
|
||||
|
||||
Print the parameter count by layer
|
||||
----------------------------------
|
||||
Whenever the .fit() function gets called, the Trainer will print the weights summary for the lightningModule.
|
||||
To disable this behavior, turn off this flag:
|
||||
|
||||
(See: :paramref:`~pytorch_lightning.trainer.trainer.Trainer.weights_summary`
|
||||
argument of :class:`~pytorch_lightning.trainer.trainer.Trainer`)
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(weights_summary=None)
|
||||
|
||||
|
||||
Set the number of validation sanity steps
|
||||
-----------------------------------------
|
||||
Lightning runs a few steps of validation in the beginning of training.
|
||||
This avoids crashing in the validation loop sometime deep into a lengthy training loop.
|
||||
|
||||
(See: :paramref:`~pytorch_lightning.trainer.trainer.Trainer.num_sanity_val_steps`
|
||||
argument of :class:`~pytorch_lightning.trainer.trainer.Trainer`)
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT
|
||||
trainer = Trainer(num_sanity_val_steps=5)
|
||||
@@ -1,8 +0,0 @@
|
||||
Documentation
|
||||
=============
|
||||
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 4
|
||||
|
||||
pytorch_lightning
|
||||
@@ -0,0 +1,66 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.callbacks.early_stopping import EarlyStopping
|
||||
|
||||
|
||||
Early stopping
|
||||
==============
|
||||
|
||||
Stopping an epoch early
|
||||
-----------------------
|
||||
You can stop an epoch early by overriding :meth:`~pytorch_lightning.core.lightning.LightningModule.on_batch_start` to return `-1` when some condition is met.
|
||||
|
||||
If you do this repeatedly, for every epoch you had originally requested, then this will stop your entire run.
|
||||
|
||||
Default Epoch End Callback Behavior
|
||||
-----------------------------------
|
||||
By default early stopping will be enabled if `'val_loss'`
|
||||
is found in :meth:`~pytorch_lightning.core.lightning.LightningModule.validation_epoch_end`'s
|
||||
return dict. Otherwise training will proceed with early stopping disabled.
|
||||
|
||||
Enable Early Stopping using Callbacks on epoch end
|
||||
--------------------------------------------------
|
||||
There are two ways to enable early stopping using callbacks on epoch end.
|
||||
|
||||
- Set early_stop_callback to True. Will look for 'val_loss' in validation_epoch_end() return dict.
|
||||
If it is not found an error is raised.
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(early_stop_callback=True)
|
||||
|
||||
- Or configure your own callback
|
||||
|
||||
.. testcode::
|
||||
|
||||
early_stop_callback = EarlyStopping(
|
||||
monitor='val_loss',
|
||||
min_delta=0.00,
|
||||
patience=3,
|
||||
verbose=False,
|
||||
mode='min'
|
||||
)
|
||||
trainer = Trainer(early_stop_callback=early_stop_callback)
|
||||
|
||||
In any case, the callback will fall back to the training metrics (returned in
|
||||
:meth:`~pytorch_lightning.core.lightning.LightningModule.training_step`,
|
||||
:meth:`~pytorch_lightning.core.lightning.LightningModule.training_step_end`)
|
||||
looking for a key to monitor if validation is disabled or
|
||||
:meth:`~pytorch_lightning.core.lightning.LightningModule.validation_epoch_end`
|
||||
is not defined.
|
||||
|
||||
.. seealso::
|
||||
- :class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
- :class:`~pytorch_lightning.callbacks.early_stopping.EarlyStopping`
|
||||
|
||||
Disable Early Stopping with callbacks on epoch end
|
||||
--------------------------------------------------
|
||||
To disable early stopping pass ``False`` to the
|
||||
:paramref:`~pytorch_lightning.trainer.trainer.Trainer.early_stop_callback`.
|
||||
Note that ``None`` will not disable early stopping but will lead to the
|
||||
default behaviour.
|
||||
|
||||
.. seealso::
|
||||
- :class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
- :class:`~pytorch_lightning.callbacks.early_stopping.EarlyStopping`
|
||||
@@ -1,8 +0,0 @@
|
||||
Examples & Tutorials
|
||||
====================
|
||||
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 3
|
||||
|
||||
pl_examples
|
||||
@@ -0,0 +1,270 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
|
||||
Experiment Logging
|
||||
==================
|
||||
|
||||
Comet.ml
|
||||
^^^^^^^^
|
||||
|
||||
`Comet.ml <https://www.comet.ml/site/>`_ is a third-party logger.
|
||||
To use :class:`~pytorch_lightning.loggers.CometLogger` as your logger do the following.
|
||||
First, install the package:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install comet-ml
|
||||
|
||||
Then configure the logger and pass it to the :class:`~pytorch_lightning.trainer.trainer.Trainer`:
|
||||
|
||||
.. testcode::
|
||||
|
||||
import os
|
||||
from pytorch_lightning.loggers import CometLogger
|
||||
comet_logger = CometLogger(
|
||||
api_key=os.environ.get('COMET_API_KEY'),
|
||||
workspace=os.environ.get('COMET_WORKSPACE'), # Optional
|
||||
save_dir='.', # Optional
|
||||
project_name='default_project', # Optional
|
||||
rest_api_key=os.environ.get('COMET_REST_API_KEY'), # Optional
|
||||
experiment_name='default' # Optional
|
||||
)
|
||||
trainer = Trainer(logger=comet_logger)
|
||||
|
||||
The :class:`~pytorch_lightning.loggers.CometLogger` is available anywhere except ``__init__`` in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def any_lightning_module_function_or_hook(self):
|
||||
some_img = fake_image()
|
||||
self.logger.experiment.add_image('generated_images', some_img, 0)
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.CometLogger` docs.
|
||||
|
||||
MLflow
|
||||
^^^^^^
|
||||
|
||||
`MLflow <https://mlflow.org/>`_ is a third-party logger.
|
||||
To use :class:`~pytorch_lightning.loggers.MLFlowLogger` as your logger do the following.
|
||||
First, install the package:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install mlflow
|
||||
|
||||
Then configure the logger and pass it to the :class:`~pytorch_lightning.trainer.trainer.Trainer`:
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import MLFlowLogger
|
||||
mlf_logger = MLFlowLogger(
|
||||
experiment_name="default",
|
||||
tracking_uri="file:./ml-runs"
|
||||
)
|
||||
trainer = Trainer(logger=mlf_logger)
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.MLFlowLogger` docs.
|
||||
|
||||
Neptune.ai
|
||||
^^^^^^^^^^
|
||||
|
||||
`Neptune.ai <https://neptune.ai/>`_ is a third-party logger.
|
||||
To use :class:`~pytorch_lightning.loggers.NeptuneLogger` as your logger do the following.
|
||||
First, install the package:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install neptune-client
|
||||
|
||||
Then configure the logger and pass it to the :class:`~pytorch_lightning.trainer.trainer.Trainer`:
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import NeptuneLogger
|
||||
neptune_logger = NeptuneLogger(
|
||||
api_key='ANONYMOUS', # replace with your own
|
||||
project_name='shared/pytorch-lightning-integration',
|
||||
experiment_name='default', # Optional,
|
||||
params={'max_epochs': 10}, # Optional,
|
||||
tags=['pytorch-lightning', 'mlp'], # Optional,
|
||||
)
|
||||
trainer = Trainer(logger=neptune_logger)
|
||||
|
||||
The :class:`~pytorch_lightning.loggers.NeptuneLogger` is available anywhere except ``__init__`` in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def any_lightning_module_function_or_hook(self):
|
||||
some_img = fake_image()
|
||||
self.logger.experiment.add_image('generated_images', some_img, 0)
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.NeptuneLogger` docs.
|
||||
|
||||
allegro.ai TRAINS
|
||||
^^^^^^^^^^^^^^^^^
|
||||
|
||||
`allegro.ai <https://github.com/allegroai/trains/>`_ is a third-party logger.
|
||||
To use :class:`~pytorch_lightning.loggers.TrainsLogger` as your logger do the following.
|
||||
First, install the package:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install trains
|
||||
|
||||
Then configure the logger and pass it to the :class:`~pytorch_lightning.trainer.trainer.Trainer`:
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import TrainsLogger
|
||||
trains_logger = TrainsLogger(
|
||||
project_name='examples',
|
||||
task_name='pytorch lightning test',
|
||||
)
|
||||
trainer = Trainer(logger=trains_logger)
|
||||
|
||||
.. testoutput::
|
||||
:options: +ELLIPSIS, +NORMALIZE_WHITESPACE
|
||||
:hide:
|
||||
|
||||
TRAINS Task: ...
|
||||
TRAINS results page: ...
|
||||
|
||||
The :class:`~pytorch_lightning.loggers.TrainsLogger` is available anywhere in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def __init__(self):
|
||||
some_img = fake_image()
|
||||
self.logger.experiment.log_image('debug', 'generated_image_0', some_img, 0)
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.TrainsLogger` docs.
|
||||
|
||||
Tensorboard
|
||||
^^^^^^^^^^^
|
||||
|
||||
To use `TensorBoard <https://pytorch.org/docs/stable/tensorboard.html>`_ as your logger do the following.
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import TensorBoardLogger
|
||||
logger = TensorBoardLogger('tb_logs', name='my_model')
|
||||
trainer = Trainer(logger=logger)
|
||||
|
||||
The :class:`~pytorch_lightning.loggers.TensorBoardLogger` is available anywhere except ``__init__`` in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def any_lightning_module_function_or_hook(self):
|
||||
some_img = fake_image()
|
||||
self.logger.experiment.add_image('generated_images', some_img, 0)
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.TensorBoardLogger` docs.
|
||||
|
||||
Test Tube
|
||||
^^^^^^^^^
|
||||
|
||||
`Test Tube <https://github.com/williamFalcon/test-tube>`_ is a
|
||||
`TensorBoard <https://pytorch.org/docs/stable/tensorboard.html>`_ logger but with nicer file structure.
|
||||
To use :class:`~pytorch_lightning.loggers.TestTubeLogger` as your logger do the following.
|
||||
First, install the package:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install test_tube
|
||||
|
||||
Then configure the logger and pass it to the :class:`~pytorch_lightning.trainer.trainer.Trainer`:
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import TestTubeLogger
|
||||
logger = TestTubeLogger('tb_logs', name='my_model')
|
||||
trainer = Trainer(logger=logger)
|
||||
|
||||
The :class:`~pytorch_lightning.loggers.TestTubeLogger` is available anywhere except ``__init__`` in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def any_lightning_module_function_or_hook(self):
|
||||
some_img = fake_image()
|
||||
self.logger.experiment.add_image('generated_images', some_img, 0)
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.TestTubeLogger` docs.
|
||||
|
||||
Weights and Biases
|
||||
^^^^^^^^^^^^^^^^^^
|
||||
|
||||
`Weights and Biases <https://www.wandb.com/>`_ is a third-party logger.
|
||||
To use :class:`~pytorch_lightning.loggers.WandbLogger` as your logger do the following.
|
||||
First, install the package:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install wandb
|
||||
|
||||
Then configure the logger and pass it to the :class:`~pytorch_lightning.trainer.trainer.Trainer`:
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import WandbLogger
|
||||
wandb_logger = WandbLogger()
|
||||
trainer = Trainer(logger=wandb_logger)
|
||||
|
||||
The :class:`~pytorch_lightning.loggers.WandbLogger` is available anywhere except ``__init__`` in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def any_lightning_module_function_or_hook(self):
|
||||
some_img = fake_image()
|
||||
self.logger.experiment.log({
|
||||
"generated_images": [wandb.Image(some_img, caption="...")]
|
||||
})
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.loggers.WandbLogger` docs.
|
||||
|
||||
Multiple Loggers
|
||||
^^^^^^^^^^^^^^^^
|
||||
|
||||
Lightning supports the use of multiple loggers, just pass a list to the
|
||||
:class:`~pytorch_lightning.trainer.trainer.Trainer`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import TensorBoardLogger, TestTubeLogger
|
||||
logger1 = TensorBoardLogger('tb_logs', name='my_model')
|
||||
logger2 = TestTubeLogger('tb_logs', name='my_model')
|
||||
trainer = Trainer(logger=[logger1, logger2])
|
||||
|
||||
The loggers are available as a list anywhere except ``__init__`` in your
|
||||
:class:`~pytorch_lightning.core.lightning.LightningModule`.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MyModule(LightningModule):
|
||||
def any_lightning_module_function_or_hook(self):
|
||||
some_img = fake_image()
|
||||
# Option 1
|
||||
self.logger.experiment[0].add_image('generated_images', some_img, 0)
|
||||
# Option 2
|
||||
self.logger[0].experiment.add_image('generated_images', some_img, 0)
|
||||
@@ -0,0 +1,127 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
|
||||
Experiment Reporting
|
||||
=====================
|
||||
|
||||
Lightning supports many different experiment loggers. These loggers allow you to monitor losses, images, text, etc...
|
||||
as training progresses. They usually provide a GUI to visualize and can sometimes even snapshot hyperparameters
|
||||
used in each experiment.
|
||||
|
||||
|
||||
Control logging frequency
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
It may slow training down to log every single batch. Trainer has an option to log every k batches instead.
|
||||
|
||||
.. testcode::
|
||||
|
||||
k = 10
|
||||
trainer = Trainer(row_log_interval=k)
|
||||
|
||||
Control log writing frequency
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Writing to a logger can be expensive. In Lightning you can set the interval at which you
|
||||
want to log using this trainer flag.
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
|
||||
.. testcode::
|
||||
|
||||
k = 100
|
||||
trainer = Trainer(log_save_interval=k)
|
||||
|
||||
Log metrics
|
||||
^^^^^^^^^^^
|
||||
|
||||
To plot metrics into whatever logger you passed in (tensorboard, comet, neptune, TRAINS, etc...)
|
||||
|
||||
1. training_epoch_end, validation_epoch_end, test_epoch_end will all log anything in the "log" key of the return dict.
|
||||
|
||||
.. testcode::
|
||||
|
||||
def training_epoch_end(self, outputs):
|
||||
loss = some_loss()
|
||||
...
|
||||
|
||||
logs = {'train_loss': loss}
|
||||
results = {'log': logs}
|
||||
return results
|
||||
|
||||
def validation_epoch_end(self, outputs):
|
||||
loss = some_loss()
|
||||
...
|
||||
|
||||
logs = {'val_loss': loss}
|
||||
results = {'log': logs}
|
||||
return results
|
||||
|
||||
def test_epoch_end(self, outputs):
|
||||
loss = some_loss()
|
||||
...
|
||||
|
||||
logs = {'test_loss': loss}
|
||||
results = {'log': logs}
|
||||
return results
|
||||
|
||||
2. In addition, you can also use any arbitrary functionality from a particular logger from within your LightningModule.
|
||||
For instance, here we log images using tensorboard.
|
||||
|
||||
.. testcode::
|
||||
:skipif: not TORCHVISION_AVAILABLE
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
self.generated_imgs = self.decoder.generate()
|
||||
|
||||
sample_imgs = self.generated_imgs[:6]
|
||||
grid = torchvision.utils.make_grid(sample_imgs)
|
||||
self.logger.experiment.add_image('generated_images', grid, 0)
|
||||
|
||||
...
|
||||
return results
|
||||
|
||||
Modify progress bar
|
||||
^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Each return dict from the training_end, validation_end, testing_end and training_step also has
|
||||
a key called "progress_bar".
|
||||
|
||||
Here we show the validation loss in the progress bar
|
||||
|
||||
.. testcode::
|
||||
|
||||
def validation_epoch_end(self, outputs):
|
||||
loss = some_loss()
|
||||
...
|
||||
|
||||
logs = {'val_loss': loss}
|
||||
results = {'progress_bar': logs}
|
||||
return results
|
||||
|
||||
Snapshot hyperparameters
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
When training a model, it's useful to know what hyperparams went into that model.
|
||||
When Lightning creates a checkpoint, it stores a key "hparams" with the hyperparams.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
lightning_checkpoint = torch.load(filepath, map_location=lambda storage, loc: storage)
|
||||
hyperparams = lightning_checkpoint['hparams']
|
||||
|
||||
Some loggers also allow logging the hyperparams used in the experiment. For instance,
|
||||
when using the TestTubeLogger or the TensorBoardLogger, all hyperparams will show
|
||||
in the `hparams tab <https://pytorch.org/docs/stable/tensorboard.html#torch.utils.tensorboard.writer.SummaryWriter.add_hparams>`_.
|
||||
|
||||
Snapshot code
|
||||
^^^^^^^^^^^^^
|
||||
Loggers also allow you to snapshot a copy of the code used in this experiment.
|
||||
For example, TestTubeLogger does this with a flag:
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.loggers import TestTubeLogger
|
||||
logger = TestTubeLogger('.', create_git_tag=True)
|
||||
@@ -0,0 +1,72 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
|
||||
Fast Training
|
||||
=============
|
||||
There are multiple options to speed up different parts of the training by choosing to train
|
||||
on a subset of data. This could be done for speed or debugging purposes.
|
||||
|
||||
Check validation every n epochs
|
||||
-------------------------------
|
||||
If you have a small dataset you might want to check validation every n epochs
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT
|
||||
trainer = Trainer(check_val_every_n_epoch=1)
|
||||
|
||||
Force training for min or max epochs
|
||||
------------------------------------
|
||||
It can be useful to force training for a minimum number of epochs or limit to a max number.
|
||||
|
||||
.. seealso::
|
||||
:class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT
|
||||
trainer = Trainer(min_epochs=1, max_epochs=1000)
|
||||
|
||||
|
||||
Set validation check frequency within 1 training epoch
|
||||
------------------------------------------------------
|
||||
For large datasets it's often desirable to check validation multiple times within a training loop.
|
||||
Pass in a float to check that often within 1 training epoch. Pass in an int k to check every k training batches.
|
||||
Must use an int if using an IterableDataset.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT
|
||||
trainer = Trainer(val_check_interval=0.95)
|
||||
|
||||
# check every .25 of an epoch
|
||||
trainer = Trainer(val_check_interval=0.25)
|
||||
|
||||
# check every 100 train batches (ie: for IterableDatasets or fixed frequency)
|
||||
trainer = Trainer(val_check_interval=100)
|
||||
|
||||
Use data subset for training, validation and test
|
||||
-------------------------------------------------
|
||||
If you don't want to check 100% of the training/validation/test set (for debugging or if it's huge), set these flags.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT
|
||||
trainer = Trainer(
|
||||
train_percent_check=1.0,
|
||||
val_percent_check=1.0,
|
||||
test_percent_check=1.0
|
||||
)
|
||||
|
||||
# check 10%, 20%, 30% only, respectively for training, validation and test set
|
||||
trainer = Trainer(
|
||||
train_percent_check=0.1,
|
||||
val_percent_check=0.2,
|
||||
test_percent_check=0.3
|
||||
)
|
||||
|
||||
.. note:: ``train_percent_check``, ``val_percent_check`` and ``test_percent_check`` will be overwritten by ``overfit_pct`` if ``overfit_pct`` > 0. ``val_percent_check`` will be ignored if ``fast_dev_run=True``.
|
||||
|
||||
.. note:: If you set ``val_percent_check=0``, validation will be disabled.
|
||||
@@ -0,0 +1,18 @@
|
||||
Pytorch Lightning Governance | Persons of interest
|
||||
==================================================
|
||||
|
||||
Leads
|
||||
-----
|
||||
- William Falcon (`williamFalcon <https://github.com/williamFalcon>`_) (Lightning founder)
|
||||
- Jirka Borovec (`Borda <https://github.com/Borda>`_)
|
||||
- Ethan Harris (`ethanwharris <https://github.com/ethanwharris>`_) (Torchbearer founder)
|
||||
- Matthew Painter (`MattPainter01 <https://github.com/MattPainter01>`_) (Torchbearer founder)
|
||||
- Justus Schock (`justusschock <https://github.com/justusschock>`_) (Former Core Member PyTorch Ignite)
|
||||
|
||||
Core Maintainers
|
||||
----------------
|
||||
- Nic Eggert (`neggert <https://github.com/neggert>`_)
|
||||
- Jeff Ling (`jeffling <https://github.com/jeffling>`_)
|
||||
- Jeremy Jordan (`jeremyjordan <https://github.com/jeremyjordan>`_)
|
||||
- Tullie Murrell (`tullie <https://github.com/tullie>`_)
|
||||
- Adrian Wälchli (`awaelchli <https://github.com/awaelchli>`_)
|
||||
@@ -0,0 +1,77 @@
|
||||
Model Hooks
|
||||
===========
|
||||
|
||||
There are cases when you might want to do something different at different parts of the training/validation loop.
|
||||
To enable a hook, simply override the method in your LightningModule and the trainer will call it at the correct time.
|
||||
|
||||
**Contributing** If there's a hook you'd like to add, simply:
|
||||
|
||||
1. Fork `PyTorchLightning <https://github.com/PyTorchLightning/pytorch-lightning>`_.
|
||||
|
||||
2. Add the hook to :class:`pytorch_lightning.core.hooks.ModelHooks`.
|
||||
|
||||
3. Add it in the correct place in :mod:`pytorch_lightning.trainer` where it should be called.
|
||||
|
||||
|
||||
Hooks lifecycle
|
||||
---------------
|
||||
|
||||
Training set-up
|
||||
^^^^^^^^^^^^^^^
|
||||
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.init_ddp_connection`
|
||||
- :meth:`~pytorch_lightning.trainer.optimizers.TrainerOptimizersMixin.init_optimizers`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.configure_apex`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.configure_ddp`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.train_dataloader`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.test_dataloader`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.val_dataloader`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.summarize`
|
||||
- :meth:`~pytorch_lightning.trainer.training_io.TrainerIOMixin.restore_weights`
|
||||
|
||||
Training loop
|
||||
^^^^^^^^^^^^^
|
||||
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_epoch_start`
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_batch_start`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.tbptt_split_batch`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.training_step`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.training_step_end` (optional)
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_before_zero_grad`
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.backward`
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_after_backward`
|
||||
- ``optimizer.step()``
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_batch_end`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.training_epoch_end`
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_epoch_end`
|
||||
|
||||
Validation loop
|
||||
^^^^^^^^^^^^^^^
|
||||
|
||||
- ``model.zero_grad()``
|
||||
- ``model.eval()``
|
||||
- ``torch.set_grad_enabled(False)``
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.validation_step`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.validation_step_end`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.validation_epoch_end`
|
||||
- ``model.train()``
|
||||
- ``torch.set_grad_enabled(True)``
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_post_performance_check`
|
||||
|
||||
Test loop
|
||||
^^^^^^^^^
|
||||
|
||||
- ``model.zero_grad()``
|
||||
- ``model.eval()``
|
||||
- ``torch.set_grad_enabled(False)``
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.test_step`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.test_step_end`
|
||||
- :meth:`~pytorch_lightning.core.lightning.LightningModule.test_epoch_end`
|
||||
- ``model.train()``
|
||||
- ``torch.set_grad_enabled(True)``
|
||||
- :meth:`~pytorch_lightning.core.hooks.ModelHooks.on_post_performance_check`
|
||||
|
||||
|
||||
|
||||
.. automodule:: pytorch_lightning.core.hooks
|
||||
:noindex:
|
||||
@@ -0,0 +1,249 @@
|
||||
.. testsetup:: *
|
||||
|
||||
import torch
|
||||
from argparse import ArgumentParser, Namespace
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
import sys
|
||||
sys.argv = ['foo']
|
||||
|
||||
|
||||
Hyperparameters
|
||||
---------------
|
||||
Lightning has utilities to interact seamlessly with the command line ArgumentParser
|
||||
and plays well with the hyperparameter optimization framework of your choice.
|
||||
|
||||
ArgumentParser
|
||||
^^^^^^^^^^^^^^
|
||||
Lightning is designed to augment a lot of the functionality of the built-in Python ArgumentParser
|
||||
|
||||
.. testcode::
|
||||
|
||||
from argparse import ArgumentParser
|
||||
parser = ArgumentParser()
|
||||
parser.add_argument('--layer_1_dim', type=int, default=128)
|
||||
args = parser.parse_args()
|
||||
|
||||
This allows you to call your program like so:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
python trainer.py --layer_1_dim 64
|
||||
|
||||
|
||||
Argparser Best Practices
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
It is best practice to layer your arguments in three sections.
|
||||
|
||||
1. Trainer args (gpus, num_nodes, etc...)
|
||||
2. Model specific arguments (layer_dim, num_layers, learning_rate, etc...)
|
||||
3. Program arguments (data_path, cluster_email, etc...)
|
||||
|
||||
We can do this as follows. First, in your LightningModule, define the arguments
|
||||
specific to that module. Remember that data splits or data paths may also be specific to
|
||||
a module (ie: if your project has a model that trains on Imagenet and another on CIFAR-10).
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
@staticmethod
|
||||
def add_model_specific_args(parent_parser):
|
||||
parser = ArgumentParser(parents=[parent_parser], add_help=False)
|
||||
parser.add_argument('--encoder_layers', type=int, default=12)
|
||||
parser.add_argument('--data_path', type=str, default='/some/path')
|
||||
return parser
|
||||
|
||||
Now in your main trainer file, add the Trainer args, the program args, and add the model args
|
||||
|
||||
.. testcode::
|
||||
|
||||
# ----------------
|
||||
# trainer_main.py
|
||||
# ----------------
|
||||
from argparse import ArgumentParser
|
||||
parser = ArgumentParser()
|
||||
|
||||
# add PROGRAM level args
|
||||
parser.add_argument('--conda_env', type=str, default='some_name')
|
||||
parser.add_argument('--notification_email', type=str, default='will@email.com')
|
||||
|
||||
# add model specific args
|
||||
parser = LitModel.add_model_specific_args(parser)
|
||||
|
||||
# add all the available trainer options to argparse
|
||||
# ie: now --gpus --num_nodes ... --fast_dev_run all work in the cli
|
||||
parser = Trainer.add_argparse_args(parser)
|
||||
|
||||
hparams = parser.parse_args()
|
||||
|
||||
Now you can call run your program like so
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
python trainer_main.py --gpus 2 --num_nodes 2 --conda_env 'my_env' --encoder_layers 12
|
||||
|
||||
Finally, make sure to start the training like so:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# YES
|
||||
model = LitModel(hparams)
|
||||
trainer = Trainer.from_argparse_args(hparams, early_stopping_callback=...)
|
||||
|
||||
# NO
|
||||
# model = LitModel(learning_rate=hparams.learning_rate, ...)
|
||||
# trainer = Trainer(gpus=hparams.gpus, ...)
|
||||
|
||||
LightningModule hparams
|
||||
^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Normally, we don't hard-code the values to a model. We usually use the command line to
|
||||
modify the network and read those values in the LightningModule
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def __init__(self, hparams):
|
||||
super().__init__()
|
||||
|
||||
# do this to save all arguments in any logger (tensorboard)
|
||||
self.hparams = hparams
|
||||
|
||||
self.layer_1 = torch.nn.Linear(28 * 28, hparams.layer_1_dim)
|
||||
self.layer_2 = torch.nn.Linear(hparams.layer_1_dim, hparams.layer_2_dim)
|
||||
self.layer_3 = torch.nn.Linear(hparams.layer_2_dim, 10)
|
||||
|
||||
def train_dataloader(self):
|
||||
return DataLoader(mnist_train, batch_size=self.hparams.batch_size)
|
||||
|
||||
def configure_optimizers(self):
|
||||
return Adam(self.parameters(), lr=self.hparams.learning_rate)
|
||||
|
||||
@staticmethod
|
||||
def add_model_specific_args(parent_parser):
|
||||
parser = ArgumentParser(parents=[parent_parser], add_help=False)
|
||||
parser.add_argument('--layer_1_dim', type=int, default=128)
|
||||
parser.add_argument('--layer_2_dim', type=int, default=256)
|
||||
parser.add_argument('--batch_size', type=int, default=64)
|
||||
parser.add_argument('--learning_rate', type=float, default=0.002)
|
||||
return parser
|
||||
|
||||
Now pass in the params when you init your model
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
parser = ArgumentParser()
|
||||
parser = LitMNIST.add_model_specific_args(parser)
|
||||
hparams = parser.parse_args()
|
||||
model = LitMNIST(hparams)
|
||||
|
||||
The line `self.hparams = hparams` is very special. This line assigns your hparams to the LightningModule.
|
||||
This does two things:
|
||||
|
||||
1. It adds them automatically to TensorBoard logs under the hparams tab.
|
||||
2. Lightning will save those hparams to the checkpoint and use them to restore the module correctly.
|
||||
|
||||
Trainer args
|
||||
^^^^^^^^^^^^
|
||||
To recap, add ALL possible trainer flags to the argparser and init the Trainer this way
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
parser = ArgumentParser()
|
||||
parser = Trainer.add_argparse_args(parser)
|
||||
hparams = parser.parse_args()
|
||||
|
||||
trainer = Trainer.from_argparse_args(hparams)
|
||||
|
||||
# or if you need to pass in callbacks
|
||||
trainer = Trainer.from_argparse_args(hparams, checkpoint_callback=..., callbacks=[...])
|
||||
|
||||
|
||||
Multiple Lightning Modules
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
We often have multiple Lightning Modules where each one has different arguments. Instead of
|
||||
polluting the main.py file, the LightningModule lets you define arguments for each one.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def __init__(self, hparams):
|
||||
super().__init__()
|
||||
self.layer_1 = torch.nn.Linear(28 * 28, hparams.layer_1_dim)
|
||||
|
||||
@staticmethod
|
||||
def add_model_specific_args(parent_parser):
|
||||
parser = ArgumentParser(parents=[parent_parser])
|
||||
parser.add_argument('--layer_1_dim', type=int, default=128)
|
||||
return parser
|
||||
|
||||
.. testcode::
|
||||
|
||||
class GoodGAN(LightningModule):
|
||||
|
||||
def __init__(self, hparams):
|
||||
super().__init__()
|
||||
self.encoder = Encoder(layers=hparams.encoder_layers)
|
||||
|
||||
@staticmethod
|
||||
def add_model_specific_args(parent_parser):
|
||||
parser = ArgumentParser(parents=[parent_parser])
|
||||
parser.add_argument('--encoder_layers', type=int, default=12)
|
||||
return parser
|
||||
|
||||
|
||||
Now we can allow each model to inject the arguments it needs in the ``main.py``
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
def main(args):
|
||||
|
||||
# pick model
|
||||
if args.model_name == 'gan':
|
||||
model = GoodGAN(hparams=args)
|
||||
elif args.model_name == 'mnist':
|
||||
model = LitMNIST(hparams=args)
|
||||
|
||||
model = LitMNIST(hparams=args)
|
||||
trainer = Trainer.from_argparse_args(args)
|
||||
trainer.fit(model)
|
||||
|
||||
if __name__ == '__main__':
|
||||
parser = ArgumentParser()
|
||||
parser = Trainer.add_argparse_args(parser)
|
||||
|
||||
# figure out which model to use
|
||||
parser.add_argument('--model_name', type=str, default='gan', help='gan or mnist')
|
||||
|
||||
# THIS LINE IS KEY TO PULL THE MODEL NAME
|
||||
temp_args, _ = parser.parse_known_args()
|
||||
|
||||
# let the model add what it wants
|
||||
if temp_args.model_name == 'gan':
|
||||
parser = GoodGAN.add_model_specific_args(parser)
|
||||
elif temp_args.model_name == 'mnist':
|
||||
parser = LitMNIST.add_model_specific_args(parser)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# train
|
||||
main(args)
|
||||
|
||||
and now we can train MNIST or the GAN using the command line interface!
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ python main.py --model_name gan --encoder_layers 24
|
||||
$ python main.py --model_name mnist --layer_1_dim 128
|
||||
|
||||
Hyperparameter Optimization
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
Lightning is fully compatible with the hyperparameter optimization libraries!
|
||||
Here are some useful ones:
|
||||
|
||||
- `Hydra <https://medium.com/pytorch/hydra-a-fresh-look-at-configuration-for-machine-learning-projects-50583186b710>`_
|
||||
- `Optuna <https://github.com/optuna/optuna/blob/master/examples/pytorch_lightning_simple.py>`_
|
||||
@@ -3,33 +3,93 @@
|
||||
You can adapt this file completely to your liking, but it should at least
|
||||
contain the root `toctree` directive.
|
||||
|
||||
Welcome to PyTorch-Lightning!
|
||||
=============================
|
||||
PyTorch Lightning Documentation
|
||||
===============================
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 4
|
||||
:maxdepth: 1
|
||||
:name: start
|
||||
:caption: Quick Start
|
||||
:caption: Start Here
|
||||
|
||||
new-project
|
||||
examples
|
||||
introduction_guide
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 4
|
||||
:maxdepth: 2
|
||||
:name: docs
|
||||
:caption: Docs
|
||||
:caption: Python API
|
||||
|
||||
documentation
|
||||
callbacks
|
||||
hooks
|
||||
lightning-module
|
||||
loggers
|
||||
trainer
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
:name: Community Examples
|
||||
:caption: Community Examples
|
||||
|
||||
Contextual Emotion Detection (DoubleDistilBert) <https://github.com/PyTorchLightning/emotion_transformer>
|
||||
Generative Adversarial Network <https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=TyYOdg8g77P0>
|
||||
Hyperparameter optimization with Optuna <https://github.com/optuna/optuna/blob/master/examples/pytorch_lightning_simple.py>
|
||||
Image Inpainting using Partial Convolutions <https://github.com/ryanwongsa/Image-Inpainting>
|
||||
MNIST on TPU <https://colab.research.google.com/drive/1-_LKx4HwAxl5M6xPJmqAAu444LTDQoa3#scrollTo=BHBz1_AnamN_>
|
||||
NER (transformers, TPU) <https://colab.research.google.com/drive/1dBN-wwYUngLYVt985wGs_OKPlK_ANB9D>
|
||||
NeuralTexture (CVPR) <https://github.com/PyTorchLightning/neuraltexture>
|
||||
Recurrent Attentive Neural Process <https://github.com/PyTorchLightning/attentive-neural-processes>
|
||||
Siamese Nets for One-shot Image Recognition <https://github.com/PyTorchLightning/Siamese-Neural-Networks>
|
||||
Speech Transformers <https://github.com/PyTorchLightning/speech-transformer-pytorch_lightning>
|
||||
Transformers transfer learning (Huggingface) <https://colab.research.google.com/drive/1F_RNcHzTfFuQf-LeKvSlud6x7jXYkG31#scrollTo=yr7eaxkF-djf>
|
||||
Transformers text classification <https://github.com/ricardorei/lightning-text-classification>
|
||||
VAE Library of over 18+ VAE flavors <https://github.com/AntixK/PyTorch-VAE>
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
:name: Tutorials
|
||||
:caption: Tutorials
|
||||
|
||||
From PyTorch to PyTorch Lightning <https://towardsdatascience.com/from-pytorch-to-pytorch-lightning-a-gentle-introduction-b371b7caaf09>
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
:name: Common Use Cases
|
||||
:caption: Common Use Cases
|
||||
|
||||
apex
|
||||
slurm
|
||||
child_modules
|
||||
debugging
|
||||
experiment_logging
|
||||
experiment_reporting
|
||||
early_stopping
|
||||
fast_training
|
||||
hooks
|
||||
hyperparameters
|
||||
lr_finder
|
||||
multi_gpu
|
||||
multiple_loaders
|
||||
weights_loading
|
||||
optimizers
|
||||
profiler
|
||||
single_gpu
|
||||
sequences
|
||||
training_tricks
|
||||
transfer_learning
|
||||
tpu
|
||||
test_set
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
:name: community
|
||||
:caption: Community
|
||||
|
||||
|
||||
CODE_OF_CONDUCT.md
|
||||
CONTRIBUTING.md
|
||||
BECOMING_A_CORE_CONTRIBUTOR.md
|
||||
|
||||
PULL_REQUEST_TEMPLATE.md
|
||||
governance.md
|
||||
|
||||
Indices and tables
|
||||
------------------
|
||||
@@ -38,3 +98,16 @@ Indices and tables
|
||||
* :ref:`modindex`
|
||||
* :ref:`search`
|
||||
|
||||
|
||||
|
||||
.. This is here to make sphinx aware of the modules but not throw an error/warning
|
||||
.. toctree::
|
||||
:hidden:
|
||||
|
||||
api/pytorch_lightning.core
|
||||
api/pytorch_lightning.callbacks
|
||||
api/pytorch_lightning.loggers
|
||||
api/pytorch_lightning.overrides
|
||||
api/pytorch_lightning.profiler
|
||||
api/pytorch_lightning.trainer
|
||||
api/pytorch_lightning.utilities
|
||||
@@ -0,0 +1,984 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
|
||||
Introduction Guide
|
||||
==================
|
||||
PyTorch Lightning provides a very simple template for organizing your PyTorch code. Once
|
||||
you've organized it into a LightningModule, it automates most of the training for you.
|
||||
|
||||
To illustrate, here's the typical PyTorch project structure organized in a LightningModule.
|
||||
|
||||
.. figure:: /_images/mnist_imgs/pt_to_pl.jpg
|
||||
:alt: Convert from PyTorch to Lightning
|
||||
|
||||
As your project grows in complexity with things like 16-bit precision, distributed training, etc... the part in blue
|
||||
quickly becomes onerous and starts distracting from the core research code.
|
||||
|
||||
---------
|
||||
|
||||
Goal of this guide
|
||||
------------------
|
||||
This guide walks through the major parts of the library to help you understand
|
||||
what each parts does. But at the end of the day, you write the same PyTorch code... just organize it
|
||||
into the LightningModule template which means you keep ALL the flexibility without having to deal with
|
||||
any of the boilerplate code
|
||||
|
||||
To show how Lightning works, we'll start with an MNIST classifier. We'll end showing how
|
||||
to use inheritance to very quickly create an AutoEncoder.
|
||||
|
||||
.. note:: Any DL/ML PyTorch project fits into the Lightning structure. Here we just focus on 3 types
|
||||
of research to illustrate.
|
||||
|
||||
---------
|
||||
|
||||
Installing Lightning
|
||||
--------------------
|
||||
Lightning is trivial to install.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
conda activate my_env
|
||||
pip install pytorch-lightning
|
||||
|
||||
Or without conda environments, anywhere you can use pip.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install pytorch-lightning
|
||||
|
||||
---------
|
||||
|
||||
Lightning Philosophy
|
||||
--------------------
|
||||
Lightning factors DL/ML code into three types:
|
||||
|
||||
- Research code
|
||||
- Engineering code
|
||||
- Non-essential code
|
||||
|
||||
Research code
|
||||
^^^^^^^^^^^^^
|
||||
In the MNIST generation example, the research code would be the particular system and how it's trained (ie: A GAN or VAE).
|
||||
In Lightning, this code is abstracted out by the `LightningModule`.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
l1 = nn.Linear(...)
|
||||
l2 = nn.Linear(...)
|
||||
decoder = Decoder()
|
||||
|
||||
x1 = l1(x)
|
||||
x2 = l2(x2)
|
||||
out = decoder(features, x)
|
||||
|
||||
loss = perceptual_loss(x1, x2, x) + CE(out, x)
|
||||
|
||||
Engineering code
|
||||
^^^^^^^^^^^^^^^^
|
||||
|
||||
The Engineering code is all the code related to training this system. Things such as early stopping, distribution
|
||||
over GPUs, 16-bit precision, etc. This is normally code that is THE SAME across most projects.
|
||||
|
||||
In Lightning, this code is abstracted out by the `Trainer`.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model.cuda(0)
|
||||
x = x.cuda(0)
|
||||
|
||||
distributed = DistributedParallel(model)
|
||||
|
||||
with gpu_zero:
|
||||
download_data()
|
||||
|
||||
dist.barrier()
|
||||
|
||||
Non-essential code
|
||||
^^^^^^^^^^^^^^^^^^
|
||||
This is code that helps the research but isn't relevant to the research code. Some examples might be:
|
||||
1. Inspect gradients
|
||||
2. Log to tensorboard.
|
||||
|
||||
In Lightning this code is abstracted out by `Callbacks`.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# log samples
|
||||
z = Q.rsample()
|
||||
generated = decoder(z)
|
||||
self.experiment.log('images', generated)
|
||||
|
||||
---------
|
||||
|
||||
Elements of a research project
|
||||
------------------------------
|
||||
Every research project requires the same core ingredients:
|
||||
|
||||
1. A model
|
||||
2. Train/val/test data
|
||||
3. Optimizer(s)
|
||||
4. Training step computations
|
||||
5. Validation step computations
|
||||
6. Test step computations
|
||||
|
||||
|
||||
The Model
|
||||
^^^^^^^^^
|
||||
The LightningModule provides the structure on how to organize these 5 ingredients.
|
||||
|
||||
Let's first start with the model. In this case we'll design
|
||||
a 3-layer neural network.
|
||||
|
||||
.. testcode::
|
||||
|
||||
import torch
|
||||
from torch.nn import functional as F
|
||||
from torch import nn
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
|
||||
# mnist images are (1, 28, 28) (channels, width, height)
|
||||
self.layer_1 = torch.nn.Linear(28 * 28, 128)
|
||||
self.layer_2 = torch.nn.Linear(128, 256)
|
||||
self.layer_3 = torch.nn.Linear(256, 10)
|
||||
|
||||
def forward(self, x):
|
||||
batch_size, channels, width, height = x.size()
|
||||
|
||||
# (b, 1, 28, 28) -> (b, 1*28*28)
|
||||
x = x.view(batch_size, -1)
|
||||
|
||||
# layer 1
|
||||
x = self.layer_1(x)
|
||||
x = torch.relu(x)
|
||||
|
||||
# layer 2
|
||||
x = self.layer_2(x)
|
||||
x = torch.relu(x)
|
||||
|
||||
# layer 3
|
||||
x = self.layer_3(x)
|
||||
|
||||
# probability distribution over labels
|
||||
x = torch.log_softmax(x, dim=1)
|
||||
|
||||
return x
|
||||
|
||||
Notice this is a `LightningModule` instead of a `torch.nn.Module`. A LightningModule is
|
||||
equivalent to a PyTorch Module except it has added functionality. However, you can use it
|
||||
EXACTLY the same as you would a PyTorch Module.
|
||||
|
||||
.. testcode::
|
||||
|
||||
net = LitMNIST()
|
||||
x = torch.Tensor(1, 1, 28, 28)
|
||||
out = net(x)
|
||||
|
||||
.. rst-class:: sphx-glr-script-out
|
||||
|
||||
Out:
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
torch.Size([1, 10])
|
||||
|
||||
Data
|
||||
^^^^
|
||||
|
||||
The Lightning Module organizes your dataloaders and data processing as well.
|
||||
Here's the PyTorch code for loading MNIST
|
||||
|
||||
.. testcode::
|
||||
:skipif: not TORCHVISION_AVAILABLE
|
||||
|
||||
from torch.utils.data import DataLoader, random_split
|
||||
from torchvision.datasets import MNIST
|
||||
import os
|
||||
from torchvision import datasets, transforms
|
||||
|
||||
# transforms
|
||||
# prepare transforms standard to MNIST
|
||||
transform=transforms.Compose([transforms.ToTensor(),
|
||||
transforms.Normalize((0.1307,), (0.3081,))])
|
||||
|
||||
# data
|
||||
mnist_train = MNIST(os.getcwd(), train=True, download=True)
|
||||
mnist_train = DataLoader(mnist_train, batch_size=64)
|
||||
|
||||
.. testoutput::
|
||||
:hide:
|
||||
:skipif: os.path.isdir(os.path.join(os.getcwd(), 'MNIST')) or not TORCHVISION_AVAILABLE
|
||||
|
||||
Downloading ...
|
||||
Extracting ...
|
||||
Downloading ...
|
||||
Extracting ...
|
||||
Downloading ...
|
||||
Extracting ...
|
||||
Processing...
|
||||
Done!
|
||||
|
||||
When using PyTorch Lightning, we use the exact same code except we organize it into
|
||||
the LightningModule
|
||||
|
||||
.. testcode::
|
||||
:skipif: not TORCHVISION_AVAILABLE
|
||||
|
||||
from torch.utils.data import DataLoader, random_split
|
||||
from torchvision.datasets import MNIST
|
||||
import os
|
||||
from torchvision import datasets, transforms
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def train_dataloader(self):
|
||||
transform=transforms.Compose([transforms.ToTensor(),
|
||||
transforms.Normalize((0.1307,), (0.3081,))])
|
||||
mnist_train = MNIST(os.getcwd(), train=True, download=False,
|
||||
transform=transform)
|
||||
return DataLoader(mnist_train, batch_size=64)
|
||||
|
||||
Notice the code is exactly the same, except now the training dataloading has been organized by the LightningModule
|
||||
under the `train_dataloader` method. This is great because if you run into a project that uses Lightning and want
|
||||
to figure out how they prepare their training data you can just look in the `train_dataloader` method.
|
||||
|
||||
Usually though, we want to separate the things that write to disk in data-processing from
|
||||
things like transforms which happen in memory.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def prepare_data(self):
|
||||
# download only
|
||||
MNIST(os.getcwd(), train=True, download=True)
|
||||
|
||||
def train_dataloader(self):
|
||||
# no download, just transform
|
||||
transform=transforms.Compose([transforms.ToTensor(),
|
||||
transforms.Normalize((0.1307,), (0.3081,))])
|
||||
mnist_train = MNIST(os.getcwd(), train=True, download=False,
|
||||
transform=transform)
|
||||
return DataLoader(mnist_train, batch_size=64)
|
||||
|
||||
Doing it in the `prepare_data` method ensures that when you have
|
||||
multiple GPUs you won't overwrite the data. This is a contrived example
|
||||
but it gets more complicated with things like NLP or Imagenet.
|
||||
|
||||
In general fill these methods with the following:
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def prepare_data(self):
|
||||
# stuff here is done once at the very beginning of training
|
||||
# before any distributed training starts
|
||||
|
||||
# download stuff
|
||||
# save to disk
|
||||
# etc...
|
||||
...
|
||||
|
||||
def train_dataloader(self):
|
||||
# data transforms
|
||||
# dataset creation
|
||||
# return a DataLoader
|
||||
...
|
||||
|
||||
Optimizer
|
||||
^^^^^^^^^
|
||||
|
||||
Next we choose what optimizer to use for training our system.
|
||||
In PyTorch we do it as follows:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from torch.optim import Adam
|
||||
optimizer = Adam(LitMNIST().parameters(), lr=1e-3)
|
||||
|
||||
|
||||
In Lightning we do the same but organize it under the configure_optimizers method.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def configure_optimizers(self):
|
||||
return Adam(self.parameters(), lr=1e-3)
|
||||
|
||||
.. note:: The LightningModule itself has the parameters, so pass in self.parameters()
|
||||
|
||||
However, if you have multiple optimizers use the matching parameters
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def configure_optimizers(self):
|
||||
return Adam(self.generator(), lr=1e-3), Adam(self.discriminator(), lr=1e-3)
|
||||
|
||||
Training step
|
||||
^^^^^^^^^^^^^
|
||||
|
||||
The training step is what happens inside the training loop.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
for epoch in epochs:
|
||||
for batch in data:
|
||||
# TRAINING STEP
|
||||
# ....
|
||||
# TRAINING STEP
|
||||
loss.backward()
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
In the case of MNIST we do the following
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
for epoch in epochs:
|
||||
for batch in data:
|
||||
# TRAINING STEP START
|
||||
x, y = batch
|
||||
logits = model(x)
|
||||
loss = F.nll_loss(logits, y)
|
||||
# TRAINING STEP END
|
||||
|
||||
loss.backward()
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
In Lightning, everything that is in the training step gets organized under the `training_step` function
|
||||
in the LightningModule
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
logits = self(x)
|
||||
loss = F.nll_loss(logits, y)
|
||||
return {'loss': loss}
|
||||
# return loss (also works)
|
||||
|
||||
Again, this is the same PyTorch code except that it has been organized by the LightningModule.
|
||||
This code is not restricted which means it can be as complicated as a full seq-2-seq, RL loop, GAN, etc...
|
||||
|
||||
---------
|
||||
|
||||
Training
|
||||
--------
|
||||
So far we defined 4 key ingredients in pure PyTorch but organized the code inside the LightningModule.
|
||||
|
||||
1. Model.
|
||||
2. Training data.
|
||||
3. Optimizer.
|
||||
4. What happens in the training loop.
|
||||
|
||||
For clarity, we'll recall that the full LightningModule now looks like this.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.layer_1 = torch.nn.Linear(28 * 28, 128)
|
||||
self.layer_2 = torch.nn.Linear(128, 256)
|
||||
self.layer_3 = torch.nn.Linear(256, 10)
|
||||
|
||||
def forward(self, x):
|
||||
batch_size, channels, width, height = x.size()
|
||||
x = x.view(batch_size, -1)
|
||||
x = self.layer_1(x)
|
||||
x = torch.relu(x)
|
||||
x = self.layer_2(x)
|
||||
x = torch.relu(x)
|
||||
x = self.layer_3(x)
|
||||
x = torch.log_softmax(x, dim=1)
|
||||
return x
|
||||
|
||||
def train_dataloader(self):
|
||||
transform=transforms.Compose([transforms.ToTensor(),
|
||||
transforms.Normalize((0.1307,), (0.3081,))])
|
||||
mnist_train = MNIST(os.getcwd(), train=True, download=False, transform=transform)
|
||||
return DataLoader(mnist_train, batch_size=64)
|
||||
|
||||
def configure_optimizers(self):
|
||||
return Adam(self.parameters(), lr=1e-3)
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
logits = self(x)
|
||||
loss = F.nll_loss(logits, y)
|
||||
|
||||
# add logging
|
||||
logs = {'loss': loss}
|
||||
return {'loss': loss, 'log': logs}
|
||||
|
||||
Again, this is the same PyTorch code, except that it's organized
|
||||
by the LightningModule. This organization now lets us train this model
|
||||
|
||||
Train on CPU
|
||||
^^^^^^^^^^^^
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from pytorch_lightning import Trainer
|
||||
|
||||
model = LitMNIST()
|
||||
trainer = Trainer()
|
||||
trainer.fit(model)
|
||||
|
||||
You should see the following weights summary and progress bar
|
||||
|
||||
.. figure:: /_images/mnist_imgs/mnist_cpu_bar.png
|
||||
:alt: mnist CPU bar
|
||||
|
||||
Logging
|
||||
^^^^^^^
|
||||
|
||||
When we added the `log` key in the return dictionary it went into the built in tensorboard logger.
|
||||
But you could have also logged by calling:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
# ...
|
||||
loss = ...
|
||||
self.logger.summary.scalar('loss', loss)
|
||||
|
||||
Which will generate automatic tensorboard logs.
|
||||
|
||||
.. figure:: /_images/mnist_imgs/mnist_tb.png
|
||||
:alt: mnist CPU bar
|
||||
|
||||
But you can also use any of the `number of other loggers <loggers.rst>`_ we support.
|
||||
|
||||
GPU training
|
||||
^^^^^^^^^^^^
|
||||
|
||||
But the beauty is all the magic you can do with the trainer flags. For instance, to run this model on a GPU:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitMNIST()
|
||||
trainer = Trainer(gpus=1)
|
||||
trainer.fit(model)
|
||||
|
||||
|
||||
.. figure:: /_images/mnist_imgs/mnist_gpu.png
|
||||
:alt: mnist GPU bar
|
||||
|
||||
Multi-GPU training
|
||||
^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Or you can also train on multiple GPUs.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitMNIST()
|
||||
trainer = Trainer(gpus=8)
|
||||
trainer.fit(model)
|
||||
|
||||
Or multiple nodes
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# (32 GPUs)
|
||||
model = LitMNIST()
|
||||
trainer = Trainer(gpus=8, num_nodes=4, distributed_backend='ddp')
|
||||
trainer.fit(model)
|
||||
|
||||
Refer to the `distributed computing guide for more details <multi_gpu.rst>`_.
|
||||
|
||||
TPUs
|
||||
^^^^
|
||||
Did you know you can use PyTorch on TPUs? It's very hard to do, but we've
|
||||
worked with the xla team to use their awesome library to get this to work
|
||||
out of the box!
|
||||
|
||||
Let's train on Colab (`full demo available here <https://colab.research.google.com/drive/1-_LKx4HwAxl5M6xPJmqAAu444LTDQoa3>`_)
|
||||
|
||||
First, change the runtime to TPU (and reinstall lightning).
|
||||
|
||||
.. figure:: /_images/mnist_imgs/runtime_tpu.png
|
||||
:alt: mnist GPU bar
|
||||
|
||||
.. figure:: /_images/mnist_imgs/restart_runtime.png
|
||||
:alt: mnist GPU bar
|
||||
|
||||
Next, install the required xla library (adds support for PyTorch on TPUs)
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import collections
|
||||
from datetime import datetime, timedelta
|
||||
import os
|
||||
import requests
|
||||
import threading
|
||||
|
||||
_VersionConfig = collections.namedtuple('_VersionConfig', 'wheels,server')
|
||||
VERSION = "torch_xla==nightly" #@param ["xrt==1.15.0", "torch_xla==nightly"]
|
||||
CONFIG = {
|
||||
'xrt==1.15.0': _VersionConfig('1.15', '1.15.0'),
|
||||
'torch_xla==nightly': _VersionConfig('nightly', 'XRT-dev{}'.format(
|
||||
(datetime.today() - timedelta(1)).strftime('%Y%m%d'))),
|
||||
}[VERSION]
|
||||
DIST_BUCKET = 'gs://tpu-pytorch/wheels'
|
||||
TORCH_WHEEL = 'torch-{}-cp36-cp36m-linux_x86_64.whl'.format(CONFIG.wheels)
|
||||
TORCH_XLA_WHEEL = 'torch_xla-{}-cp36-cp36m-linux_x86_64.whl'.format(CONFIG.wheels)
|
||||
TORCHVISION_WHEEL = 'torchvision-{}-cp36-cp36m-linux_x86_64.whl'.format(CONFIG.wheels)
|
||||
|
||||
# Update TPU XRT version
|
||||
def update_server_xrt():
|
||||
print('Updating server-side XRT to {} ...'.format(CONFIG.server))
|
||||
url = 'http://{TPU_ADDRESS}:8475/requestversion/{XRT_VERSION}'.format(
|
||||
TPU_ADDRESS=os.environ['COLAB_TPU_ADDR'].split(':')[0],
|
||||
XRT_VERSION=CONFIG.server,
|
||||
)
|
||||
print('Done updating server-side XRT: {}'.format(requests.post(url)))
|
||||
|
||||
update = threading.Thread(target=update_server_xrt)
|
||||
update.start()
|
||||
|
||||
.. code-block::
|
||||
|
||||
# Install Colab TPU compat PyTorch/TPU wheels and dependencies
|
||||
!pip uninstall -y torch torchvision
|
||||
!gsutil cp "$DIST_BUCKET/$TORCH_WHEEL" .
|
||||
!gsutil cp "$DIST_BUCKET/$TORCH_XLA_WHEEL" .
|
||||
!gsutil cp "$DIST_BUCKET/$TORCHVISION_WHEEL" .
|
||||
!pip install "$TORCH_WHEEL"
|
||||
!pip install "$TORCH_XLA_WHEEL"
|
||||
!pip install "$TORCHVISION_WHEEL"
|
||||
!sudo apt-get install libomp5
|
||||
update.join()
|
||||
|
||||
In distributed training (multiple GPUs and multiple TPU cores) each GPU or TPU core will run a copy
|
||||
of this program. This means that without taking any care you will download the dataset N times which
|
||||
will cause all sorts of issues.
|
||||
|
||||
To solve this problem, move the download code to the `prepare_data` method in the LightningModule.
|
||||
In this method we do all the preparation we need to do once (instead of on every gpu).
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
def prepare_data(self):
|
||||
# transform
|
||||
transform=transforms.Compose([transforms.ToTensor(), transforms.Normalize((0.1307,), (0.3081,))])
|
||||
|
||||
# download
|
||||
mnist_train = MNIST(os.getcwd(), train=True, download=True, transform=transform)
|
||||
mnist_test = MNIST(os.getcwd(), train=False, download=True, transform=transform)
|
||||
|
||||
# train/val split
|
||||
mnist_train, mnist_val = random_split(mnist_train, [55000, 5000])
|
||||
|
||||
# assign to use in dataloaders
|
||||
self.train_dataset = mnist_train
|
||||
self.val_dataset = mnist_val
|
||||
self.test_dataset = mnist_test
|
||||
|
||||
def train_dataloader(self):
|
||||
return DataLoader(self.train_dataset, batch_size=64)
|
||||
|
||||
def val_dataloader(self):
|
||||
return DataLoader(self.val_dataset, batch_size=64)
|
||||
|
||||
def test_dataloader(self):
|
||||
return DataLoader(self.test_dataset, batch_size=64)
|
||||
|
||||
The `prepare_data` method is also a good place to do any data processing that needs to be done only
|
||||
once (ie: download or tokenize, etc...).
|
||||
|
||||
.. note:: Lightning inserts the correct DistributedSampler for distributed training. No need to add yourself!
|
||||
|
||||
Now we can train the LightningModule on a TPU without doing anything else!
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitMNIST()
|
||||
trainer = Trainer(num_tpu_cores=8)
|
||||
trainer.fit(model)
|
||||
|
||||
You'll now see the TPU cores booting up.
|
||||
|
||||
.. figure:: /_images/mnist_imgs/tpu_start.png
|
||||
:alt: TPU start
|
||||
|
||||
Notice the epoch is MUCH faster!
|
||||
|
||||
.. figure:: /_images/mnist_imgs/tpu_fast.png
|
||||
:alt: TPU speed
|
||||
|
||||
---------
|
||||
|
||||
.. include:: hyperparameters.rst
|
||||
|
||||
---------
|
||||
|
||||
Validating
|
||||
----------
|
||||
|
||||
For most cases, we stop training the model when the performance on a validation
|
||||
split of the data reaches a minimum.
|
||||
|
||||
Just like the `training_step`, we can define a `validation_step` to check whatever
|
||||
metrics we care about, generate samples or add more to our logs.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
for epoch in epochs:
|
||||
for batch in data:
|
||||
# ...
|
||||
# train
|
||||
|
||||
# validate
|
||||
outputs = []
|
||||
for batch in val_data:
|
||||
x, y = batch # validation_step
|
||||
y_hat = model(x) # validation_step
|
||||
loss = loss(y_hat, x) # validation_step
|
||||
outputs.append({'val_loss': loss}) # validation_step
|
||||
|
||||
full_loss = outputs.mean() # validation_epoch_end
|
||||
|
||||
Since the `validation_step` processes a single batch,
|
||||
in Lightning we also have a `validation_epoch_end` method which allows you to compute
|
||||
statistics on the full dataset after an epoch of validation data and not just the batch.
|
||||
|
||||
In addition, we define a `val_dataloader` method which tells the trainer what data to use for validation.
|
||||
Notice we split the train split of MNIST into train, validation. We also have to make sure to do the
|
||||
sample split in the `train_dataloader` method.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
def validation_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
logits = self(x)
|
||||
loss = F.nll_loss(logits, y)
|
||||
return {'val_loss': loss}
|
||||
|
||||
def validation_epoch_end(self, outputs):
|
||||
avg_loss = torch.stack([x['val_loss'] for x in outputs]).mean()
|
||||
tensorboard_logs = {'val_loss': avg_loss}
|
||||
return {'val_loss': avg_loss, 'log': tensorboard_logs}
|
||||
|
||||
def val_dataloader(self):
|
||||
transform=transforms.Compose([transforms.ToTensor(),
|
||||
transforms.Normalize((0.1307,), (0.3081,))])
|
||||
mnist_train = MNIST(os.getcwd(), train=True, download=False,
|
||||
transform=transform)
|
||||
_, mnist_val = random_split(mnist_train, [55000, 5000])
|
||||
mnist_val = DataLoader(mnist_val, batch_size=64)
|
||||
return mnist_val
|
||||
|
||||
Again, we've just organized the regular PyTorch code into two steps, the `validation_step` method which
|
||||
operates on a single batch and the `validation_epoch_end` method to compute statistics on all batches.
|
||||
|
||||
If you have these methods defined, Lightning will call them automatically. Now we can train
|
||||
while checking the validation set.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from pytorch_lightning import Trainer
|
||||
|
||||
model = LitMNIST()
|
||||
trainer = Trainer(num_tpu_cores=8)
|
||||
trainer.fit(model)
|
||||
|
||||
You may have noticed the words `Validation sanity check` logged. This is because Lightning runs 5 batches
|
||||
of validation before starting to train. This is a kind of unit test to make sure that if you have a bug
|
||||
in the validation loop, you won't need to potentially wait a full epoch to find out.
|
||||
|
||||
.. note:: Lightning disables gradients, puts model in eval mode and does everything needed for validation.
|
||||
|
||||
---------
|
||||
|
||||
Testing
|
||||
-------
|
||||
Once our research is done and we're about to publish or deploy a model, we normally want to figure out
|
||||
how it will generalize in the "real world." For this, we use a held-out split of the data for testing.
|
||||
|
||||
Just like the validation loop, we define exactly the same steps for testing:
|
||||
|
||||
- test_step
|
||||
- test_epoch_end
|
||||
- test_dataloader
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
def test_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
logits = self(x)
|
||||
loss = F.nll_loss(logits, y)
|
||||
return {'val_loss': loss}
|
||||
|
||||
def test_epoch_end(self, outputs):
|
||||
avg_loss = torch.stack([x['val_loss'] for x in outputs]).mean()
|
||||
tensorboard_logs = {'val_loss': avg_loss}
|
||||
return {'val_loss': avg_loss, 'log': tensorboard_logs}
|
||||
|
||||
def test_dataloader(self):
|
||||
transform=transforms.Compose([transforms.ToTensor(), transforms.Normalize((0.1307,), (0.3081,))])
|
||||
mnist_train = MNIST(os.getcwd(), train=False, download=False, transform=transform)
|
||||
_, mnist_val = random_split(mnist_train, [55000, 5000])
|
||||
mnist_val = DataLoader(mnist_val, batch_size=64)
|
||||
return mnist_val
|
||||
|
||||
However, to make sure the test set isn't used inadvertently, Lightning has a separate API to run tests.
|
||||
Once you train your model simply call `.test()`.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from pytorch_lightning import Trainer
|
||||
|
||||
model = LitMNIST()
|
||||
trainer = Trainer(num_tpu_cores=8)
|
||||
trainer.fit(model)
|
||||
|
||||
# run test set
|
||||
trainer.test()
|
||||
|
||||
.. rst-class:: sphx-glr-script-out
|
||||
|
||||
Out:
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
--------------------------------------------------------------
|
||||
TEST RESULTS
|
||||
{'test_loss': tensor(1.1703, device='cuda:0')}
|
||||
--------------------------------------------------------------
|
||||
|
||||
You can also run the test from a saved lightning model
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitMNIST.load_from_checkpoint(PATH)
|
||||
trainer = Trainer(num_tpu_cores=8)
|
||||
trainer.test(model)
|
||||
|
||||
.. note:: Lightning disables gradients, puts model in eval mode and does everything needed for testing.
|
||||
|
||||
.. warning:: .test() is not stable yet on TPUs. We're working on getting around the multiprocessing challenges.
|
||||
|
||||
---------
|
||||
|
||||
Predicting
|
||||
----------
|
||||
Again, a LightningModule is exactly the same as a PyTorch module. This means you can load it
|
||||
and use it for prediction.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitMNIST.load_from_checkpoint(PATH)
|
||||
x = torch.Tensor(1, 1, 28, 28)
|
||||
out = model(x)
|
||||
|
||||
On the surface, it looks like `forward` and `training_step` are similar. Generally, we want to make sure that
|
||||
what we want the model to do is what happens in the `forward`. whereas the `training_step` likely calls forward from
|
||||
within it.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MNISTClassifier(LightningModule):
|
||||
|
||||
def forward(self, x):
|
||||
batch_size, channels, width, height = x.size()
|
||||
x = x.view(batch_size, -1)
|
||||
x = self.layer_1(x)
|
||||
x = torch.relu(x)
|
||||
x = self.layer_2(x)
|
||||
x = torch.relu(x)
|
||||
x = self.layer_3(x)
|
||||
x = torch.log_softmax(x, dim=1)
|
||||
return x
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
logits = self(x)
|
||||
loss = F.nll_loss(logits, y)
|
||||
return loss
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = MNISTClassifier()
|
||||
x = mnist_image()
|
||||
logits = model(x)
|
||||
|
||||
In this case, we've set this LightningModel to predict logits. But we could also have it predict feature maps:
|
||||
|
||||
.. testcode::
|
||||
|
||||
class MNISTRepresentator(LightningModule):
|
||||
|
||||
def forward(self, x):
|
||||
batch_size, channels, width, height = x.size()
|
||||
x = x.view(batch_size, -1)
|
||||
x = self.layer_1(x)
|
||||
x1 = torch.relu(x)
|
||||
x = self.layer_2(x1)
|
||||
x2 = torch.relu(x)
|
||||
x3 = self.layer_3(x2)
|
||||
return [x, x1, x2, x3]
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
out, l1_feats, l2_feats, l3_feats = self(x)
|
||||
logits = torch.log_softmax(out, dim=1)
|
||||
ce_loss = F.nll_loss(logits, y)
|
||||
loss = perceptual_loss(l1_feats, l2_feats, l3_feats) + ce_loss
|
||||
return loss
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = MNISTRepresentator.load_from_checkpoint(PATH)
|
||||
x = mnist_image()
|
||||
feature_maps = model(x)
|
||||
|
||||
Or maybe we have a model that we use to do generation
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNISTDreamer(LightningModule):
|
||||
|
||||
def forward(self, z):
|
||||
imgs = self.decoder(z)
|
||||
return imgs
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
representation = self.encoder(x)
|
||||
imgs = self(representation)
|
||||
|
||||
loss = perceptual_loss(imgs, x)
|
||||
return loss
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitMNISTDreamer.load_from_checkpoint(PATH)
|
||||
z = sample_noise()
|
||||
generated_imgs = model(z)
|
||||
|
||||
How you split up what goes in `forward` vs `training_step` depends on how you want to use this model for
|
||||
prediction.
|
||||
|
||||
---------
|
||||
|
||||
Extensibility
|
||||
-------------
|
||||
Although lightning makes everything super simple, it doesn't sacrifice any flexibility or control.
|
||||
Lightning offers multiple ways of managing the training state.
|
||||
|
||||
Training overrides
|
||||
^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Any part of the training, validation and testing loop can be modified.
|
||||
For instance, if you wanted to do your own backward pass, you would override the
|
||||
default implementation
|
||||
|
||||
.. testcode::
|
||||
|
||||
def backward(self, use_amp, loss, optimizer):
|
||||
if use_amp:
|
||||
with amp.scale_loss(loss, optimizer) as scaled_loss:
|
||||
scaled_loss.backward()
|
||||
else:
|
||||
loss.backward()
|
||||
|
||||
With your own
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def backward(self, use_amp, loss, optimizer):
|
||||
# do a custom way of backward
|
||||
loss.backward(retain_graph=True)
|
||||
|
||||
Or if you wanted to initialize ddp in a different way than the default one
|
||||
|
||||
.. testcode::
|
||||
|
||||
def configure_ddp(self, model, device_ids):
|
||||
# Lightning DDP simply routes to test_step, val_step, etc...
|
||||
model = LightningDistributedDataParallel(
|
||||
model,
|
||||
device_ids=device_ids,
|
||||
find_unused_parameters=True
|
||||
)
|
||||
return model
|
||||
|
||||
you could do your own:
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitMNIST(LightningModule):
|
||||
|
||||
def configure_ddp(self, model, device_ids):
|
||||
|
||||
model = Horovod(model)
|
||||
# model = Ray(model)
|
||||
return model
|
||||
|
||||
Every single part of training is configurable this way.
|
||||
For a full list look at `LightningModule <lightning-module.rst>`_.
|
||||
|
||||
---------
|
||||
|
||||
Callbacks
|
||||
---------
|
||||
Another way to add arbitrary functionality is to add a custom callback
|
||||
for hooks that you might care about
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.callbacks import Callback
|
||||
|
||||
class MyPrintingCallback(Callback):
|
||||
|
||||
def on_init_start(self, trainer):
|
||||
print('Starting to init trainer!')
|
||||
|
||||
def on_init_end(self, trainer):
|
||||
print('Trainer is init now')
|
||||
|
||||
def on_train_end(self, trainer, pl_module):
|
||||
print('do something when training ends')
|
||||
|
||||
And pass the callbacks into the trainer
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(callbacks=[MyPrintingCallback()])
|
||||
|
||||
.. testoutput::
|
||||
:hide:
|
||||
|
||||
Starting to init trainer!
|
||||
Trainer is init now
|
||||
|
||||
.. note::
|
||||
See full list of 12+ hooks in the :ref:`callbacks`.
|
||||
|
||||
---------
|
||||
|
||||
.. include:: child_modules.rst
|
||||
|
||||
---------
|
||||
|
||||
.. include:: transfer_learning.rst
|
||||
@@ -0,0 +1,12 @@
|
||||
.. role:: hidden
|
||||
:class: hidden-section
|
||||
|
||||
LightningModule
|
||||
===============
|
||||
|
||||
.. automodule:: pytorch_lightning.core
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_abc_impl,
|
||||
summarize,
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
.. role:: hidden
|
||||
:class: hidden-section
|
||||
|
||||
Loggers
|
||||
===========
|
||||
.. automodule:: pytorch_lightning.loggers
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_abc_impl,
|
||||
_save_model,
|
||||
on_epoch_end,
|
||||
on_train_end,
|
||||
on_epoch_start,
|
||||
@@ -0,0 +1,116 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
Learning Rate Finder
|
||||
--------------------
|
||||
|
||||
For training deep neural networks, selecting a good learning rate is essential
|
||||
for both better performance and faster convergence. Even optimizers such as
|
||||
`Adam` that are self-adjusting the learning rate can benefit from more optimal
|
||||
choices.
|
||||
|
||||
To reduce the amount of guesswork concerning choosing a good initial learning
|
||||
rate, a `learning rate finder` can be used. As described in this `paper <https://arxiv.org/abs/1506.01186>`_
|
||||
a learning rate finder does a small run where the learning rate is increased
|
||||
after each processed batch and the corresponding loss is logged. The result of
|
||||
this is a `lr` vs. `loss` plot that can be used as guidance for choosing a optimal
|
||||
initial lr.
|
||||
|
||||
Warnings:
|
||||
- For the moment, this feature only works with models having a single optimizer.
|
||||
- LR support for DDP is not implemented yet, it is comming soon.
|
||||
|
||||
Using Lightnings build-in LR finder
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
In the most basic use case, this feature can be enabled during trainer construction
|
||||
with ``Trainer(auto_lr_find=True)``. When ``.fit(model)`` is called, the lr finder
|
||||
will automatically be run before any training is done. The ``lr`` that is found
|
||||
and used will be written to the console and logged together with all other
|
||||
hyperparameters of the model.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# default, no automatic learning rate finder
|
||||
trainer = Trainer(auto_lr_find=True)
|
||||
|
||||
When the ``lr`` or ``learning_rate`` key in hparams exists, this flag sets your learning_rate.
|
||||
In both cases, if the respective fields are not found, an error will be thrown.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def __init__(self, hparams):
|
||||
self.hparams = hparams
|
||||
|
||||
def configure_optimizers(self):
|
||||
return Adam(self.parameters(), lr=self.hparams.lr|self.hparams.learning_rate)
|
||||
|
||||
# finds learning rate automatically
|
||||
# sets hparams.lr or hparams.learning_rate to that learning rate
|
||||
trainer = Trainer(auto_lr_find=True)
|
||||
|
||||
To use an arbitrary value set it in the parameter.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# to set to your own hparams.my_value
|
||||
trainer = Trainer(auto_lr_find='my_value')
|
||||
|
||||
Under the hood, when you call fit, this is what happens.
|
||||
|
||||
1. Run learning rate finder.
|
||||
2. Run actual fit.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# when you call .fit() this happens
|
||||
# 1. find learning rate
|
||||
# 2. actually run fit
|
||||
trainer.fit(model)
|
||||
|
||||
If you want to inspect the results of the learning rate finder before doing any
|
||||
actual training or just play around with the parameters of the algorithm, this
|
||||
can be done by invoking the ``lr_find`` method of the trainer. A typical example
|
||||
of this would look like
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = MyModelClass(hparams)
|
||||
trainer = Trainer()
|
||||
|
||||
# Run learning rate finder
|
||||
lr_finder = trainer.lr_find(model)
|
||||
|
||||
# Results can be found in
|
||||
lr_finder.results
|
||||
|
||||
# Plot with
|
||||
fig = lr_finder.plot(suggest=True)
|
||||
fig.show()
|
||||
|
||||
# Pick point based on plot, or get suggestion
|
||||
new_lr = lr_finder.suggestion()
|
||||
|
||||
# update hparams of the model
|
||||
model.hparams.lr = new_lr
|
||||
|
||||
# Fit model
|
||||
trainer.fit(model)
|
||||
|
||||
The figure produced by ``lr_finder.plot()`` should look something like the figure
|
||||
below. It is recommended to not pick the learning rate that achives the lowest
|
||||
loss, but instead something in the middle of the sharpest downward slope (red point).
|
||||
This is the point returned py ``lr_finder.suggestion()``.
|
||||
|
||||
.. figure:: /_images/trainer/lr_finder.png
|
||||
|
||||
The parameters of the algorithm can be seen below.
|
||||
|
||||
.. autoclass:: pytorch_lightning.trainer.lr_finder.TrainerLRFinderMixin
|
||||
:members: lr_find
|
||||
:noindex:
|
||||
:exclude-members: _run_lr_finder_internally, save_checkpoint, restore
|
||||
@@ -0,0 +1,414 @@
|
||||
.. testsetup:: *
|
||||
|
||||
import torch
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
.. _multi-gpu-training:
|
||||
|
||||
Multi-GPU training
|
||||
==================
|
||||
Lightning supports multiple ways of doing distributed training.
|
||||
|
||||
Preparing your code
|
||||
-------------------
|
||||
To train on CPU/GPU/TPU without changing your code, we need to build a few good habits :)
|
||||
|
||||
Delete .cuda() or .to() calls
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Delete any calls to .cuda() or .to(device).
|
||||
|
||||
.. testcode::
|
||||
|
||||
# before lightning
|
||||
def forward(self, x):
|
||||
x = x.cuda(0)
|
||||
layer_1.cuda(0)
|
||||
x_hat = layer_1(x)
|
||||
|
||||
# after lightning
|
||||
def forward(self, x):
|
||||
x_hat = layer_1(x)
|
||||
|
||||
Init using type_as
|
||||
^^^^^^^^^^^^^^^^^^
|
||||
When you need to create a new tensor, use `type_as`.
|
||||
This will make your code scale to any arbitrary number of GPUs or TPUs with Lightning
|
||||
|
||||
.. testcode::
|
||||
|
||||
# before lightning
|
||||
def forward(self, x):
|
||||
z = torch.Tensor(2, 3)
|
||||
z = z.cuda(0)
|
||||
|
||||
# with lightning
|
||||
def forward(self, x):
|
||||
z = torch.Tensor(2, 3)
|
||||
z = z.type_as(x, device=self.device)
|
||||
|
||||
Every LightningModule knows what device it is on. You can access that reference via `self.device`.
|
||||
|
||||
Remove samplers
|
||||
^^^^^^^^^^^^^^^
|
||||
For multi-node or TPU training, in PyTorch we must use `torch.nn.DistributedSampler`. The
|
||||
sampler makes sure each GPU sees the appropriate part of your data.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# without lightning
|
||||
def train_dataloader(self):
|
||||
dataset = MNIST(...)
|
||||
sampler = None
|
||||
|
||||
if self.on_tpu:
|
||||
sampler = DistributedSampler(dataset)
|
||||
|
||||
return DataLoader(dataset, sampler=sampler)
|
||||
|
||||
With Lightning, you don't need to do this because it takes care of adding the correct samplers
|
||||
when needed.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# with lightning
|
||||
def train_dataloader(self):
|
||||
dataset = MNIST(...)
|
||||
return DataLoader(dataset)
|
||||
|
||||
.. note:: If you don't want this behavior, disable it with `Trainer(replace_sampler_ddp=False)`
|
||||
|
||||
.. note:: For iterable datasets, we don't do this automatically.
|
||||
|
||||
Make Model Picklable
|
||||
^^^^^^^^^^^^^^^^^^^^
|
||||
It's very likely your code is already `picklable <https://docs.python.org/3/library/pickle.html>`_,
|
||||
so you don't have to do anything to make this change.
|
||||
However, if you run distributed and see an error like this:
|
||||
|
||||
.. code-block::
|
||||
|
||||
self._launch(process_obj)
|
||||
File "/net/software/local/python/3.6.5/lib/python3.6/multiprocessing/popen_spawn_posix.py", line 47,
|
||||
in _launch reduction.dump(process_obj, fp)
|
||||
File "/net/software/local/python/3.6.5/lib/python3.6/multiprocessing/reduction.py", line 60, in dump
|
||||
ForkingPickler(file, protocol).dump(obj)
|
||||
_pickle.PicklingError: Can't pickle <function <lambda> at 0x2b599e088ae8>:
|
||||
attribute lookup <lambda> on __main__ failed
|
||||
|
||||
This means you have something in your model definition, transforms, optimizer, dataloader or callbacks
|
||||
that is cannot be pickled. By pickled we mean the following would fail.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import pickle
|
||||
pickle.dump(some_object)
|
||||
|
||||
This is a limitation of using multiple processes for distributed training within PyTorch.
|
||||
To fix this issue, find your piece of code that cannot be pickled. The end of the stacktrace
|
||||
is usually helpful.
|
||||
|
||||
.. code-block::
|
||||
|
||||
self._launch(process_obj)
|
||||
File "/net/software/local/python/3.6.5/lib/python3.6/multiprocessing/popen_spawn_posix.py", line 47,
|
||||
in _launch reduction.dump(process_obj, fp)
|
||||
File "/net/software/local/python/3.6.5/lib/python3.6/multiprocessing/reduction.py", line 60, in dump
|
||||
ForkingPickler(file, protocol).dump(obj)
|
||||
_pickle.PicklingError: Can't pickle [THIS IS THE THING TO FIND AND DELETE]:
|
||||
attribute lookup <lambda> on __main__ failed
|
||||
|
||||
ie: in the stacktrace example here, there seems to be a lambda function somewhere in the user code
|
||||
which cannot be pickled.
|
||||
|
||||
Distributed modes
|
||||
-----------------
|
||||
Lightning allows multiple ways of training
|
||||
|
||||
- Data Parallel (`distributed_backend='dp'`) (multiple-gpus, 1 machine)
|
||||
- DistributedDataParallel (`distributed_backend='ddp'`) (multiple-gpus across many machines).
|
||||
- DistributedDataParallel2 (`distributed_backend='ddp2'`) (dp in a machine, ddp across machines).
|
||||
- Horovod (`distributed_backend='horovod'`) (multi-machine, multi-gpu, configured at runtime)
|
||||
- TPUs (`num_tpu_cores=8|x`) (tpu or TPU pod)
|
||||
|
||||
.. note:: If you request multiple GPUs without setting a mode, ddp will be automatically used.
|
||||
|
||||
Data Parallel (dp)
|
||||
^^^^^^^^^^^^^^^^^^
|
||||
`DataParallel <https://pytorch.org/docs/stable/nn.html#torch.nn.DataParallel>`_ splits a batch across k GPUs. That is, if you have a batch of 32 and use dp with 2 gpus,
|
||||
each GPU will process 16 samples, after which the root node will aggregate the results.
|
||||
|
||||
.. warning:: DP use is discouraged by PyTorch and Lightning. Use ddp which is more stable and at least 3x faster
|
||||
|
||||
.. testcode::
|
||||
:skipif: torch.cuda.device_count() < 2
|
||||
|
||||
# train on 2 GPUs (using dp mode)
|
||||
trainer = Trainer(gpus=2, distributed_backend='dp')
|
||||
|
||||
Distributed Data Parallel
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
`DistributedDataParallel <https://pytorch.org/docs/stable/nn.html#distributeddataparallel>`_ works as follows.
|
||||
|
||||
1. Each GPU across every node gets its own process.
|
||||
|
||||
2. Each GPU gets visibility into a subset of the overall dataset. It will only ever see that subset.
|
||||
|
||||
3. Each process inits the model.
|
||||
|
||||
.. note:: Make sure to set the random seed so that each model inits with the same weights
|
||||
|
||||
4. Each process performs a full forward and backward pass in parallel.
|
||||
|
||||
5. The gradients are synced and averaged across all processes.
|
||||
|
||||
6. Each process updates its optimizer.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# train on 8 GPUs (same machine (ie: node))
|
||||
trainer = Trainer(gpus=8, distributed_backend='ddp')
|
||||
|
||||
# train on 32 GPUs (4 nodes)
|
||||
trainer = Trainer(gpus=8, distributed_backend='ddp', num_nodes=4)
|
||||
|
||||
Distributed Data Parallel 2
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
In certain cases, it's advantageous to use all batches on the same machine instead of a subset.
|
||||
For instance you might want to compute a NCE loss where it pays to have more negative samples.
|
||||
|
||||
In this case, we can use ddp2 which behaves like dp in a machine and ddp across nodes. DDP2 does the following:
|
||||
|
||||
1. Copies a subset of the data to each node.
|
||||
|
||||
2. Inits a model on each node.
|
||||
|
||||
3. Runs a forward and backward pass using DP.
|
||||
|
||||
4. Syncs gradients across nodes.
|
||||
|
||||
5. Applies the optimizer updates.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# train on 32 GPUs (4 nodes)
|
||||
trainer = Trainer(gpus=8, distributed_backend='ddp2', num_nodes=4)
|
||||
|
||||
Horovod
|
||||
^^^^^^^
|
||||
`Horovod <http://horovod.ai>`_ allows the same training script to be used for single-GPU,
|
||||
multi-GPU, and multi-node training.
|
||||
|
||||
Like Distributed Data Parallel, every process in Horovod operates on a single GPU with a fixed
|
||||
subset of the data. Gradients are averaged across all GPUs in parallel during the backward pass,
|
||||
then synchronously applied before beginning the next step.
|
||||
|
||||
The number of worker processes is configured by a driver application (`horovodrun` or `mpirun`). In
|
||||
the training script, Horovod will detect the number of workers from the environment, and automatically
|
||||
scale the learning rate to compensate for the increased total batch size.
|
||||
|
||||
Horovod can be configured in the training script to run with any number of GPUs / processes as follows:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# train Horovod on GPU (number of GPUs / machines provided on command-line)
|
||||
trainer = Trainer(distributed_backend='horovod', gpus=1)
|
||||
|
||||
# train Horovod on CPU (number of processes / machines provided on command-line)
|
||||
trainer = Trainer(distributed_backend='horovod')
|
||||
|
||||
When starting the training job, the driver application will then be used to specify the total
|
||||
number of worker processes:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# run training with 4 GPUs on a single machine
|
||||
horovodrun -np 4 python train.py
|
||||
|
||||
# run training with 8 GPUs on two machines (4 GPUs each)
|
||||
horovodrun -np 8 -H hostname1:4,hostname2:4 python train.py
|
||||
|
||||
See the official `Horovod documentation <https://horovod.readthedocs.io/en/stable>`_ for details
|
||||
on installation and performance tuning.
|
||||
|
||||
DP/DDP2 caveats
|
||||
^^^^^^^^^^^^^^^
|
||||
In DP and DDP2 each GPU within a machine sees a portion of a batch.
|
||||
DP and ddp2 roughly do the following:
|
||||
|
||||
.. testcode::
|
||||
|
||||
def distributed_forward(batch, model):
|
||||
batch = torch.Tensor(32, 8)
|
||||
gpu_0_batch = batch[:8]
|
||||
gpu_1_batch = batch[8:16]
|
||||
gpu_2_batch = batch[16:24]
|
||||
gpu_3_batch = batch[24:]
|
||||
|
||||
y_0 = model_copy_gpu_0(gpu_0_batch)
|
||||
y_1 = model_copy_gpu_1(gpu_1_batch)
|
||||
y_2 = model_copy_gpu_2(gpu_2_batch)
|
||||
y_3 = model_copy_gpu_3(gpu_3_batch)
|
||||
|
||||
return [y_0, y_1, y_2, y_3]
|
||||
|
||||
So, when Lightning calls any of the `training_step`, `validation_step`, `test_step`
|
||||
you will only be operating on one of those pieces.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# the batch here is a portion of the FULL batch
|
||||
def training_step(self, batch, batch_idx):
|
||||
y_0 = batch
|
||||
|
||||
For most metrics, this doesn't really matter. However, if you want
|
||||
to add something to your computational graph (like softmax)
|
||||
using all batch parts you can use the `training_step_end` step.
|
||||
|
||||
.. testcode::
|
||||
|
||||
def training_step_end(self, outputs):
|
||||
# only use when on dp
|
||||
outputs = torch.cat(outputs, dim=1)
|
||||
softmax = softmax(outputs, dim=1)
|
||||
out = softmax.mean()
|
||||
return out
|
||||
|
||||
In pseudocode, the full sequence is:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# get data
|
||||
batch = next(dataloader)
|
||||
|
||||
# copy model and data to each gpu
|
||||
batch_splits = split_batch(batch, num_gpus)
|
||||
models = copy_model_to_gpus(model)
|
||||
|
||||
# in parallel, operate on each batch chunk
|
||||
all_results = []
|
||||
for gpu_num in gpus:
|
||||
batch_split = batch_splits[gpu_num]
|
||||
gpu_model = models[gpu_num]
|
||||
out = gpu_model(batch_split)
|
||||
all_results.append(out)
|
||||
|
||||
# use the full batch for something like softmax
|
||||
full out = model.training_step_end(all_results)
|
||||
|
||||
to illustrate why this is needed, let's look at dataparallel
|
||||
|
||||
.. testcode::
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
y_hat = self(batch)
|
||||
|
||||
# on dp or ddp2 if we did softmax now it would be wrong
|
||||
# because batch is actually a piece of the full batch
|
||||
return y_hat
|
||||
|
||||
def training_step_end(self, batch_parts_outputs):
|
||||
# batch_parts_outputs has outputs of each part of the batch
|
||||
|
||||
# do softmax here
|
||||
outputs = torch.cat(outputs, dim=1)
|
||||
softmax = softmax(outputs, dim=1)
|
||||
out = softmax.mean()
|
||||
|
||||
return out
|
||||
|
||||
If `training_step_end` is defined it will be called regardless of tpu, dp, ddp, etc... which means
|
||||
it will behave the same no matter the backend.
|
||||
|
||||
Validation and test step also have the same option when using dp
|
||||
|
||||
.. testcode::
|
||||
|
||||
def validation_step_end(self, batch_parts_outputs):
|
||||
...
|
||||
|
||||
def test_step_end(self, batch_parts_outputs):
|
||||
...
|
||||
|
||||
Implement Your Own Distributed (DDP) training
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
If you need your own way to init PyTorch DDP you can override :meth:`pytorch_lightning.core.LightningModule.`.
|
||||
|
||||
If you also need to use your own DDP implementation, override: :meth:`pytorch_lightning.core.LightningModule.configure_ddp`.
|
||||
|
||||
|
||||
Batch size
|
||||
----------
|
||||
When using distributed training make sure to modify your learning rate according to your effective
|
||||
batch size.
|
||||
|
||||
Let's say you have a batch size of 7 in your dataloader.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def train_dataloader(self):
|
||||
return Dataset(..., batch_size=7)
|
||||
|
||||
In (DDP, Horovod) your effective batch size will be 7 * gpus * num_nodes.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# effective batch size = 7 * 8
|
||||
Trainer(gpus=8, distributed_backend='ddp|horovod')
|
||||
|
||||
# effective batch size = 7 * 8 * 10
|
||||
Trainer(gpus=8, num_nodes=10, distributed_backend='ddp|horovod')
|
||||
|
||||
|
||||
In DDP2, your effective batch size will be 7 * num_nodes.
|
||||
The reason is that the full batch is visible to all GPUs on the node when using DDP2.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# effective batch size = 7
|
||||
Trainer(gpus=8, distributed_backend='ddp2')
|
||||
|
||||
# effective batch size = 7 * 10
|
||||
Trainer(gpus=8, num_nodes=10, distributed_backend='ddp2')
|
||||
|
||||
|
||||
.. note:: Huge batch sizes are actually really bad for convergence. Check out:
|
||||
`Accurate, Large Minibatch SGD: Training ImageNet in 1 Hour <https://arxiv.org/abs/1706.02677>`_
|
||||
|
||||
PytorchElastic
|
||||
--------------
|
||||
Lightning supports the use of PytorchElastic to enable fault-tolerent and elastic distributed job scheduling. To use it, specify the 'ddp' or 'ddp2' backend and the number of gpus you want to use in the trainer.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
Trainer(gpus=8, distributed_backend='ddp')
|
||||
|
||||
|
||||
Following the `PytorchElastic Quickstart documentation <https://pytorch.org/elastic/0.2.0/quickstart.html>`_, you then need to start a single-node etcd server on one of the hosts:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
etcd --enable-v2
|
||||
--listen-client-urls http://0.0.0.0:2379,http://127.0.0.1:4001
|
||||
--advertise-client-urls PUBLIC_HOSTNAME:2379
|
||||
|
||||
|
||||
And then launch the elastic job with:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
python -m torchelastic.distributed.launch
|
||||
--nnodes=MIN_SIZE:MAX_SIZE
|
||||
--nproc_per_node=TRAINERS_PER_NODE
|
||||
--rdzv_id=JOB_ID
|
||||
--rdzv_backend=etcd
|
||||
--rdzv_endpoint=ETCD_HOST:ETCD_PORT
|
||||
YOUR_LIGHTNING_TRAINING_SCRIPT.py (--arg1 ... train script args...)
|
||||
|
||||
|
||||
See the official `PytorchElastic documentation <https://pytorch.org/elastic/0.2.0/index.html>`_ for details
|
||||
on installation and more use cases.
|
||||
@@ -0,0 +1,73 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
Multiple Datasets
|
||||
=================
|
||||
Lightning supports multiple dataloaders in a few ways.
|
||||
|
||||
1. Create a dataloader that iterates both datasets under the hood.
|
||||
2. In the validation and test loop you also have the option to return multiple dataloaders
|
||||
which lightning will call sequentially.
|
||||
|
||||
Multiple training dataloaders
|
||||
-----------------------------
|
||||
For training, the best way to use multiple-dataloaders is to create a Dataloader class
|
||||
which wraps both your dataloaders. (This of course also works for testing and validation
|
||||
dataloaders).
|
||||
|
||||
(`reference <https://discuss.pytorch.org/t/train-simultaneously-on-two-datasets/649/2>`_)
|
||||
|
||||
.. testcode::
|
||||
|
||||
class ConcatDataset(torch.utils.data.Dataset):
|
||||
def __init__(self, *datasets):
|
||||
self.datasets = datasets
|
||||
|
||||
def __getitem__(self, i):
|
||||
return tuple(d[i] for d in self.datasets)
|
||||
|
||||
def __len__(self):
|
||||
return min(len(d) for d in self.datasets)
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def train_dataloader(self):
|
||||
concat_dataset = ConcatDataset(
|
||||
datasets.ImageFolder(traindir_A),
|
||||
datasets.ImageFolder(traindir_B)
|
||||
)
|
||||
|
||||
loader = torch.utils.data.DataLoader(
|
||||
concat_dataset,
|
||||
batch_size=args.batch_size,
|
||||
shuffle=True,
|
||||
num_workers=args.workers,
|
||||
pin_memory=True
|
||||
)
|
||||
return loader
|
||||
|
||||
def val_dataloader(self):
|
||||
# SAME
|
||||
...
|
||||
|
||||
def test_dataloader(self):
|
||||
# SAME
|
||||
...
|
||||
|
||||
Test/Val dataloaders
|
||||
--------------------
|
||||
For validation, test dataloaders lightning also gives you the additional
|
||||
option of passing in multiple dataloaders back from each call.
|
||||
|
||||
See the following for more details:
|
||||
|
||||
- :meth:`~pytorch_lightning.core.LightningModule.val_dataloader`
|
||||
- :meth:`~pytorch_lightning.core.LightningModule.test_dataloader`
|
||||
|
||||
.. testcode::
|
||||
|
||||
def val_dataloader(self):
|
||||
loader_1 = Dataloader()
|
||||
loader_2 = Dataloader()
|
||||
return [loader_1, loader_2]
|
||||
@@ -1,71 +1,279 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
|
||||
|
||||
Quick Start
|
||||
===========
|
||||
To start a new project define two files, a LightningModule and a Trainer file.
|
||||
To illustrate Lightning power and simplicity, here's an example of a typical research flow.
|
||||
|
||||
Case 1: BERT
|
||||
------------
|
||||
PyTorch Lightning is nothing more than organized PyTorch code.
|
||||
Once you've organized it into a LightningModule, it automates most of the training for you.
|
||||
|
||||
Let's say you're working on something like BERT but want to try different ways of training or even different networks.
|
||||
You would define a single LightningModule and use flags to switch between your different ideas.
|
||||
To illustrate, here's the typical PyTorch project structure organized in a LightningModule.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
class BERT(pl.LightningModule):
|
||||
def __init__(self, model_name, task):
|
||||
self.task = task
|
||||
|
||||
if model_name == 'transformer':
|
||||
self.net = Transformer()
|
||||
elif model_name == 'my_cool_version':
|
||||
self.net = MyCoolVersion()
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
if self.task == 'standard_bert':
|
||||
# do standard bert training with self.net...
|
||||
# return loss
|
||||
|
||||
if self.task == 'my_cool_task':
|
||||
# do my own version with self.net
|
||||
# return loss
|
||||
.. figure:: /_images/mnist_imgs/pt_to_pl.jpg
|
||||
:alt: Convert from PyTorch to Lightning
|
||||
|
||||
|
||||
Case 2: COOLER NOT BERT
|
||||
-----------------------
|
||||
Step 1: Define a LightningModule
|
||||
---------------------------------
|
||||
|
||||
But if you wanted to try something **completely** different, you'd define a new module for that.
|
||||
.. testcode::
|
||||
:skipif: not TORCHVISION_AVAILABLE
|
||||
|
||||
import os
|
||||
|
||||
.. code-block:: python
|
||||
import torch
|
||||
from torch.nn import functional as F
|
||||
from torch.utils.data import DataLoader
|
||||
from torchvision.datasets import MNIST
|
||||
from torchvision import transforms
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
class CoolerNotBERT(pl.LightningModule):
|
||||
def __init__(self):
|
||||
self.net = ...
|
||||
super().__init__()
|
||||
self.l1 = torch.nn.Linear(28 * 28, 10)
|
||||
|
||||
def forward(self, x):
|
||||
return torch.relu(self.l1(x.view(x.size(0), -1)))
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
# do some other cool task
|
||||
# return loss
|
||||
x, y = batch
|
||||
y_hat = self(x)
|
||||
loss = F.cross_entropy(y_hat, y)
|
||||
tensorboard_logs = {'train_loss': loss}
|
||||
return {'loss': loss, 'log': tensorboard_logs}
|
||||
|
||||
def configure_optimizers(self):
|
||||
return torch.optim.Adam(self.parameters(), lr=0.001)
|
||||
|
||||
def train_dataloader(self):
|
||||
dataset = MNIST(os.getcwd(), train=True, download=True, transform=transforms.ToTensor())
|
||||
loader = DataLoader(dataset, batch_size=32, num_workers=4, shuffle=True)
|
||||
return loader
|
||||
|
||||
|
||||
Rapid research flow
|
||||
-------------------
|
||||
Step 2: Fit with a Trainer
|
||||
--------------------------
|
||||
|
||||
Then you could do rapid research by switching between these two and using the same trainer.
|
||||
.. testcode::
|
||||
:skipif: torch.cuda.device_count() < 8
|
||||
|
||||
from pytorch_lightning import Trainer
|
||||
|
||||
.. code-block:: python
|
||||
model = LitModel()
|
||||
|
||||
if use_bert:
|
||||
model = BERT()
|
||||
else:
|
||||
model = CoolerNotBERT()
|
||||
|
||||
trainer = Trainer(gpus=4, use_amp=True)
|
||||
# most basic trainer, uses good defaults
|
||||
trainer = Trainer(gpus=8, num_nodes=1)
|
||||
trainer.fit(model)
|
||||
|
||||
Under the hood, lightning does (in high-level pseudocode):
|
||||
|
||||
**Notice a few things about this flow:**
|
||||
.. code-block:: python
|
||||
|
||||
1. You're writing pure PyTorch... no unnecessary abstractions or new libraries to learn.
|
||||
2. You get free GPU and 16-bit support without writing any of that code in your model.
|
||||
3. You also get all of the capabilities below (without coding or testing yourself).
|
||||
model = LitModel()
|
||||
train_dataloader = model.train_dataloader()
|
||||
optimizer = model.configure_optimizers()
|
||||
|
||||
for epoch in epochs:
|
||||
train_outs = []
|
||||
for batch in train_dataloader:
|
||||
loss = model.training_step(batch)
|
||||
loss.backward()
|
||||
train_outs.append(loss.detach())
|
||||
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
# optional for logging, etc...
|
||||
model.training_epoch_end(train_outs)
|
||||
|
||||
Validation loop
|
||||
---------------
|
||||
To also add a validation loop add the following functions
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def validation_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
y_hat = self(x)
|
||||
return {'val_loss': F.cross_entropy(y_hat, y)}
|
||||
|
||||
def validation_epoch_end(self, outputs):
|
||||
avg_loss = torch.stack([x['val_loss'] for x in outputs]).mean()
|
||||
tensorboard_logs = {'val_loss': avg_loss}
|
||||
return {'val_loss': avg_loss, 'log': tensorboard_logs}
|
||||
|
||||
def val_dataloader(self):
|
||||
# TODO: do a real train/val split
|
||||
dataset = MNIST(os.getcwd(), train=False, download=True, transform=transforms.ToTensor())
|
||||
loader = DataLoader(dataset, batch_size=32, num_workers=4)
|
||||
return loader
|
||||
|
||||
And now the trainer will call the validation loop automatically
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# most basic trainer, uses good defaults
|
||||
trainer = Trainer(gpus=8, num_nodes=1)
|
||||
trainer.fit(model)
|
||||
|
||||
Under the hood in pseudocode, lightning does the following:
|
||||
|
||||
.. testsetup:: *
|
||||
|
||||
train_dataloader = []
|
||||
|
||||
.. testcode::
|
||||
|
||||
# ...
|
||||
for batch in train_dataloader:
|
||||
loss = model.training_step()
|
||||
loss.backward()
|
||||
# ...
|
||||
|
||||
if validate_at_some_point:
|
||||
model.eval()
|
||||
val_outs = []
|
||||
for val_batch in model.val_dataloader:
|
||||
val_out = model.validation_step(val_batch)
|
||||
val_outs.append(val_out)
|
||||
|
||||
model.validation_epoch_end(val_outs)
|
||||
model.train()
|
||||
|
||||
The beauty of Lightning is that it handles the details of when to validate, when to call .eval(),
|
||||
turning off gradients, detaching graphs, making sure you don't enable shuffle for val, etc...
|
||||
|
||||
.. note:: Lightning removes all the million details you need to remember during research
|
||||
|
||||
Test loop
|
||||
---------
|
||||
You might also need a test loop
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def test_step(self, batch, batch_idx):
|
||||
x, y = batch
|
||||
y_hat = self(x)
|
||||
return {'test_loss': F.cross_entropy(y_hat, y)}
|
||||
|
||||
def test_epoch_end(self, outputs):
|
||||
avg_loss = torch.stack([x['test_loss'] for x in outputs]).mean()
|
||||
tensorboard_logs = {'test_loss': avg_loss}
|
||||
return {'avg_test_loss': avg_loss, 'log': tensorboard_logs}
|
||||
|
||||
def test_dataloader(self):
|
||||
# TODO: do a real train/val split
|
||||
dataset = MNIST(os.getcwd(), train=False, download=True, transform=transforms.ToTensor())
|
||||
loader = DataLoader(dataset, batch_size=32, num_workers=4)
|
||||
return loader
|
||||
|
||||
However, this time you need to specifically call test (this is done so you don't use the test set by mistake)
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# OPTION 1:
|
||||
# test after fit
|
||||
trainer.fit(model)
|
||||
trainer.test()
|
||||
|
||||
# OPTION 2:
|
||||
# test after loading weights
|
||||
model = LitModel.load_from_checkpoint(PATH)
|
||||
trainer = Trainer(num_tpu_cores=1)
|
||||
trainer.test()
|
||||
|
||||
Again, under the hood, lightning does the following in (pseudocode):
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model.eval()
|
||||
test_outs = []
|
||||
for test_batch in model.test_dataloader:
|
||||
test_out = model.test_step(val_batch)
|
||||
test_outs.append(test_out)
|
||||
|
||||
model.test_epoch_end(test_outs)
|
||||
|
||||
Datasets
|
||||
--------
|
||||
If you don't want to define the datasets as part of the LightningModule, just pass them into fit instead.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# pass in datasets if you want.
|
||||
train_dataloader = DataLoader(dataset, batch_size=32, num_workers=4)
|
||||
val_dataloader, test_dataloader = ...
|
||||
|
||||
trainer = Trainer(gpus=8, num_nodes=1)
|
||||
trainer.fit(model, train_dataloader, val_dataloader)
|
||||
|
||||
trainer.test(test_dataloader=test_dataloader)
|
||||
|
||||
The advantage of this method is the ability to reuse models for different datasets. The disadvantage
|
||||
is that for research it makes readability and reproducibility more difficult. This is why we recommend
|
||||
to define the datasets in the LightningModule if you're doing research, but use the method above for
|
||||
production models or for prediction tasks.
|
||||
|
||||
Why do you need Lightning?
|
||||
--------------------------
|
||||
Notice the code above has nothing about .cuda() or 16-bit or early stopping or logging, etc...
|
||||
This is where Lightning adds a ton of value.
|
||||
|
||||
Without changing a SINGLE line of your code, you can now do the following with the above code
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# train on TPUs using 16 bit precision with early stopping
|
||||
# using only half the training data and checking validation every quarter of a training epoch
|
||||
trainer = Trainer(
|
||||
nb_tpu_cores=8,
|
||||
precision=16,
|
||||
early_stop_checkpoint=True,
|
||||
train_percent_check=0.5,
|
||||
val_check_interval=0.25
|
||||
)
|
||||
|
||||
# train on 256 GPUs
|
||||
trainer = Trainer(
|
||||
gpus=8,
|
||||
num_nodes=32
|
||||
)
|
||||
|
||||
# train on 1024 CPUs across 128 machines
|
||||
trainer = Trainer(
|
||||
num_processes=8,
|
||||
num_nodes=128
|
||||
)
|
||||
|
||||
And the best part is that your code is STILL just PyTorch... meaning you can do anything you
|
||||
would normally do.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitModel()
|
||||
model.eval()
|
||||
|
||||
y_hat = model(x)
|
||||
|
||||
model.anything_you_can_do_with_pytorch()
|
||||
|
||||
Summary
|
||||
-------
|
||||
In short, by refactoring your PyTorch code:
|
||||
|
||||
1. You STILL keep pure PyTorch.
|
||||
2. You DON't lose any flexibility.
|
||||
3. You can get rid of all of your boilerplate.
|
||||
4. You make your code generalizable to any hardware.
|
||||
5. Your code is now readable and easier to reproduce (ie: you help with the reproducibility crisis).
|
||||
6. Your LightningModule is still just a pure PyTorch module.
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
Optimization
|
||||
===============
|
||||
|
||||
Learning rate scheduling
|
||||
-------------------------------------
|
||||
Every optimizer you use can be paired with any `LearningRateScheduler <https://pytorch.org/docs/stable/optim.html#how-to-adjust-learning-rate>`_.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# no LR scheduler
|
||||
def configure_optimizers(self):
|
||||
return Adam(...)
|
||||
|
||||
# Adam + LR scheduler
|
||||
def configure_optimizers(self):
|
||||
optimizer = Adam(...)
|
||||
scheduler = ReduceLROnPlateau(optimizer, ...)
|
||||
return [optimizer], [scheduler]
|
||||
|
||||
# Two optimziers each with a scheduler
|
||||
def configure_optimizers(self):
|
||||
optimizer1 = Adam(...)
|
||||
optimizer2 = SGD(...)
|
||||
scheduler1 = ReduceLROnPlateau(optimizer1, ...)
|
||||
scheduler2 = LambdaLR(optimizer2, ...)
|
||||
return [optimizer1, optimizer2], [scheduler1, scheduler2]
|
||||
|
||||
# Same as above with additional params passed to the first scheduler
|
||||
def configure_optimizers(self):
|
||||
optimizers = [Adam(...), SGD(...)]
|
||||
schedulers = [
|
||||
{
|
||||
'scheduler': ReduceLROnPlateau(optimizers[0], ...),
|
||||
'monitor': 'val_recall', # Default: val_loss
|
||||
'interval': 'epoch',
|
||||
'frequency': 1
|
||||
},
|
||||
LambdaLR(optimizers[1], ...)
|
||||
]
|
||||
return optimizers, schedulers
|
||||
|
||||
|
||||
Use multiple optimizers (like GANs)
|
||||
-------------------------------------
|
||||
To use multiple optimizers return > 1 optimizers from :meth:`pytorch_lightning.core.LightningModule.configure_optimizers`
|
||||
|
||||
.. testcode::
|
||||
|
||||
# one optimizer
|
||||
def configure_optimizers(self):
|
||||
return Adam(...)
|
||||
|
||||
# two optimizers, no schedulers
|
||||
def configure_optimizers(self):
|
||||
return Adam(...), SGD(...)
|
||||
|
||||
# Two optimizers, one scheduler for adam only
|
||||
def configure_optimizers(self):
|
||||
return [Adam(...), SGD(...)], [ReduceLROnPlateau()]
|
||||
|
||||
Lightning will call each optimizer sequentially:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
for epoch in epochs:
|
||||
for batch in data:
|
||||
for opt in optimizers:
|
||||
train_step(opt)
|
||||
opt.step()
|
||||
|
||||
for scheduler in scheduler:
|
||||
scheduler.step()
|
||||
|
||||
|
||||
Step optimizers at arbitrary intervals
|
||||
----------------------------------------
|
||||
To do more interesting things with your optimizers such as learning rate warm-up or odd scheduling,
|
||||
override the :meth:`optimizer_step` function.
|
||||
|
||||
For example, here step optimizer A every 2 batches and optimizer B every 4 batches
|
||||
|
||||
.. testcode::
|
||||
|
||||
def optimizer_step(self, current_epoch, batch_nb, optimizer, optimizer_i, second_order_closure=None):
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
# Alternating schedule for optimizer steps (ie: GANs)
|
||||
def optimizer_step(self, current_epoch, batch_nb, optimizer, optimizer_i, second_order_closure=None):
|
||||
# update generator opt every 2 steps
|
||||
if optimizer_i == 0:
|
||||
if batch_nb % 2 == 0 :
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
# update discriminator opt every 4 steps
|
||||
if optimizer_i == 1:
|
||||
if batch_nb % 4 == 0 :
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
# ...
|
||||
# add as many optimizers as you want
|
||||
|
||||
Here we add a learning-rate warm up
|
||||
|
||||
.. testcode::
|
||||
|
||||
# learning rate warm-up
|
||||
def optimizer_step(self, current_epoch, batch_nb, optimizer, optimizer_i, second_order_closure=None):
|
||||
# warm up lr
|
||||
if self.trainer.global_step < 500:
|
||||
lr_scale = min(1., float(self.trainer.global_step + 1) / 500.)
|
||||
for pg in optimizer.param_groups:
|
||||
pg['lr'] = lr_scale * self.hparams.learning_rate
|
||||
|
||||
# update params
|
||||
optimizer.step()
|
||||
optimizer.zero_grad()
|
||||
@@ -0,0 +1,11 @@
|
||||
.. role:: hidden
|
||||
:class: hidden-section
|
||||
|
||||
|
||||
Performance and Bottleneck Profiler
|
||||
===================================
|
||||
.. automodule:: pytorch_lightning.profiler
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
_abc_impl,
|
||||
summarize,
|
||||
@@ -0,0 +1,84 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from torch.utils.data import IterableDataset
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
Sequential Data
|
||||
================
|
||||
Lightning has built in support for dealing with sequential data.
|
||||
|
||||
|
||||
Packed sequences as inputs
|
||||
----------------------------
|
||||
When using PackedSequence, do 2 things:
|
||||
|
||||
1. return either a padded tensor in dataset or a list of variable length tensors in the dataloader collate_fn (example above shows the list implementation).
|
||||
2. Pack the sequence in forward or training and validation steps depending on use case.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# For use in dataloader
|
||||
def collate_fn(batch):
|
||||
x = [item[0] for item in batch]
|
||||
y = [item[1] for item in batch]
|
||||
return x, y
|
||||
|
||||
# In module
|
||||
def training_step(self, batch, batch_nb):
|
||||
x = rnn.pack_sequence(batch[0], enforce_sorted=False)
|
||||
y = rnn.pack_sequence(batch[1], enforce_sorted=False)
|
||||
|
||||
Truncated Backpropagation Through Time
|
||||
---------------------------------------
|
||||
There are times when multiple backwards passes are needed for each batch.
|
||||
For example, it may save memory to use Truncated Backpropagation Through Time when training RNNs.
|
||||
|
||||
Lightning can handle TBTT automatically via this flag.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT (single backwards pass per batch)
|
||||
trainer = Trainer(truncated_bptt_steps=None)
|
||||
|
||||
# (split batch into sequences of size 2)
|
||||
trainer = Trainer(truncated_bptt_steps=2)
|
||||
|
||||
.. note:: If you need to modify how the batch is split,
|
||||
override :meth:`pytorch_lightning.core.LightningModule.tbptt_split_batch`.
|
||||
|
||||
.. note:: Using this feature requires updating your LightningModule's :meth:`pytorch_lightning.core.LightningModule.training_step` to include
|
||||
a `hiddens` arg.
|
||||
|
||||
Iterable Datasets
|
||||
---------------------------------------
|
||||
Lightning supports using IterableDatasets as well as map-style Datasets. IterableDatasets provide a more natural
|
||||
option when using sequential data.
|
||||
|
||||
.. note:: When using an IterableDataset you must set the val_check_interval to 1.0 (the default) or to an int
|
||||
(specifying the number of training batches to run before validation) when initializing the Trainer.
|
||||
This is due to the fact that the IterableDataset does not have a __len__ and Lightning requires this to calculate
|
||||
the validation interval when val_check_interval is less than one.
|
||||
|
||||
.. testcode::
|
||||
|
||||
# IterableDataset
|
||||
class CustomDataset(IterableDataset):
|
||||
|
||||
def __init__(self, data):
|
||||
self.data_source
|
||||
|
||||
def __iter__(self):
|
||||
return iter(self.data_source)
|
||||
|
||||
# Setup DataLoader
|
||||
def train_dataloader(self):
|
||||
seq_data = ['A', 'long', 'time', 'ago', 'in', 'a', 'galaxy', 'far', 'far', 'away']
|
||||
iterable_dataset = CustomDataset(seq_data)
|
||||
|
||||
dataloader = DataLoader(dataset=iterable_dataset, batch_size=5)
|
||||
return dataloader
|
||||
|
||||
.. testcode::
|
||||
|
||||
# Set val_check_interval
|
||||
trainer = Trainer(val_check_interval=100)
|
||||
@@ -0,0 +1,14 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
Single GPU Training
|
||||
====================
|
||||
Make sure you are running on a machine that has at least one GPU. Lightning handles all the NVIDIA flags for you,
|
||||
there's no need to set them yourself.
|
||||
|
||||
.. testcode::
|
||||
:skipif: torch.cuda.device_count() < 1
|
||||
|
||||
# train on 1 GPU (using dp mode)
|
||||
trainer = Trainer(gpus=1)
|
||||
@@ -0,0 +1,111 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
Computing cluster (SLURM)
|
||||
=========================
|
||||
|
||||
Lightning automates job the details behind training on a SLURM powered cluster.
|
||||
|
||||
.. _multi-node:
|
||||
|
||||
Multi-node training
|
||||
-------------------
|
||||
To train a model using multiple-nodes do the following:
|
||||
|
||||
1. Design your LightningModule.
|
||||
|
||||
2. Enable ddp in the trainer
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# train on 32 GPUs across 4 nodes
|
||||
trainer = Trainer(gpus=8, num_nodes=4, distributed_backend='ddp')
|
||||
|
||||
3. It's a good idea to structure your train.py file like this:
|
||||
|
||||
.. testcode::
|
||||
|
||||
# train.py
|
||||
def main(hparams):
|
||||
model = LightningTemplateModel(hparams)
|
||||
|
||||
trainer = pl.Trainer(
|
||||
gpus=8,
|
||||
num_nodes=4,
|
||||
distributed_backend='ddp'
|
||||
)
|
||||
|
||||
trainer.fit(model)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
root_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
parent_parser = ArgumentParser(add_help=False)
|
||||
hyperparams = parser.parse_args()
|
||||
|
||||
# TRAIN
|
||||
main(hyperparams)
|
||||
|
||||
4. Create the appropriate SLURM job
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# (submit.sh)
|
||||
#!/bin/bash -l
|
||||
|
||||
# SLURM SUBMIT SCRIPT
|
||||
#SBATCH --nodes=4
|
||||
#SBATCH --gres=gpu:8
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --mem=0
|
||||
#SBATCH --time=0-02:00:00
|
||||
|
||||
# activate conda env
|
||||
source activate $1
|
||||
|
||||
# -------------------------
|
||||
# debugging flags (optional)
|
||||
export NCCL_DEBUG=INFO
|
||||
export PYTHONFAULTHANDLER=1
|
||||
|
||||
# on your cluster you might need these:
|
||||
# set the network interface
|
||||
# export NCCL_SOCKET_IFNAME=^docker0,lo
|
||||
|
||||
# might need the latest cuda
|
||||
# module load NCCL/2.4.7-1-cuda.10.0
|
||||
# -------------------------
|
||||
|
||||
# run script from above
|
||||
srun python3 train.py
|
||||
|
||||
5. If you want auto-resubmit (read below), add this line to the submit.sh script
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
#SBATCH --signal=SIGUSR1@90
|
||||
|
||||
6. Submit the SLURM job
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sbatch submit.sh
|
||||
|
||||
.. note:: using :class:`~torch.utils.data.distributed.DistributedSampler` is already handled by Lightning.
|
||||
|
||||
Walltime auto-resubmit
|
||||
----------------------
|
||||
When you use Lightning in a SLURM cluster, lightning automatically detects when it is about
|
||||
to run into the walltime, and it does the following:
|
||||
|
||||
1. Saves a temporary checkpoint.
|
||||
2. Requeues the job.
|
||||
3. When the job starts, it loads the temporary checkpoint.
|
||||
|
||||
To get this behavior make sure to add the correct signal to your SLURM script
|
||||
|
||||
.. code-block::
|
||||
|
||||
# 90 seconds before training ends
|
||||
#SBATCH --signal=SIGUSR1@90
|
||||
@@ -0,0 +1,37 @@
|
||||
Test set
|
||||
========
|
||||
Lightning forces the user to run the test set separately to make sure it isn't evaluated by mistake
|
||||
|
||||
|
||||
Test after fit
|
||||
--------------
|
||||
To run the test set after training completes, use this method
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# run full training
|
||||
trainer.fit(model)
|
||||
|
||||
# run test set
|
||||
trainer.test()
|
||||
|
||||
Test pre-trained model
|
||||
----------------------
|
||||
To run the test set on a pre-trained model, use this method.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = MyLightningModule.load_from_checkpoint(
|
||||
checkpoint_path='/path/to/pytorch_checkpoint.ckpt',
|
||||
hparams_file='/path/to/test_tube/experiment/version/hparams.yaml',
|
||||
map_location=None
|
||||
)
|
||||
|
||||
# init trainer with whatever options
|
||||
trainer = Trainer(...)
|
||||
|
||||
# test (pass in the model)
|
||||
trainer.test(model)
|
||||
|
||||
In this case, the options you pass to trainer will be used when
|
||||
running the test set (ie: 16-bit, dp, ddp, etc...)
|
||||
@@ -0,0 +1,210 @@
|
||||
TPU support
|
||||
===========
|
||||
|
||||
Lightning supports running on TPUs. At this moment, TPUs are only available
|
||||
on Google Cloud (GCP). For more information on TPUs
|
||||
`watch this video <https://www.youtube.com/watch?v=kPMpmcl_Pyw>`_.
|
||||
|
||||
---------------
|
||||
|
||||
Live demo
|
||||
----------
|
||||
Check out this `Google Colab <https://colab.research.google.com/drive/1-_LKx4HwAxl5M6xPJmqAAu444LTDQoa3>`_ to see how to train MNIST on TPUs.
|
||||
|
||||
---------------
|
||||
|
||||
TPU Terminology
|
||||
---------------
|
||||
A TPU is a Tensor processing unit. Each TPU has 8 cores where each
|
||||
core is optimized for 128x128 matrix multiplies. In general, a single
|
||||
TPU is about as fast as 5 V100 GPUs!
|
||||
|
||||
A TPU pod hosts many TPUs on it. Currently, TPU pod v2 has 2048 cores!
|
||||
You can request a full pod from Google cloud or a "slice" which gives you
|
||||
some subset of those 2048 cores.
|
||||
|
||||
---------------
|
||||
|
||||
How to access TPUs
|
||||
-------------------
|
||||
To access TPUs there are two main ways.
|
||||
|
||||
1. Using google colab.
|
||||
2. Using Google Cloud (GCP).
|
||||
|
||||
---------------
|
||||
|
||||
Colab TPUs
|
||||
-----------
|
||||
Colab is like a jupyter notebook with a free GPU or TPU
|
||||
hosted on GCP.
|
||||
|
||||
To get a TPU on colab, follow these steps:
|
||||
|
||||
1. Go to `https://colab.research.google.com/ <https://colab.research.google.com/>`_.
|
||||
|
||||
2. Click "new notebook" (bottom right of pop-up).
|
||||
|
||||
3. Click runtime > change runtime settings. Select Python 3, and hardware accelerator "TPU".
|
||||
This will give you a TPU with 8 cores.
|
||||
|
||||
4. Next, insert this code into the first cell and execute.
|
||||
This will install the xla library that interfaces between PyTorch and the TPU.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import collections
|
||||
from datetime import datetime, timedelta
|
||||
import os
|
||||
import requests
|
||||
import threading
|
||||
|
||||
_VersionConfig = collections.namedtuple('_VersionConfig', 'wheels,server')
|
||||
VERSION = "xrt==1.15.0" #@param ["xrt==1.15.0", "torch_xla==nightly"]
|
||||
CONFIG = {
|
||||
'xrt==1.15.0': _VersionConfig('1.15', '1.15.0'),
|
||||
'torch_xla==nightly': _VersionConfig('nightly', 'XRT-dev{}'.format(
|
||||
(datetime.today() - timedelta(1)).strftime('%Y%m%d'))),
|
||||
}[VERSION]
|
||||
DIST_BUCKET = 'gs://tpu-pytorch/wheels'
|
||||
TORCH_WHEEL = 'torch-{}-cp36-cp36m-linux_x86_64.whl'.format(CONFIG.wheels)
|
||||
TORCH_XLA_WHEEL = 'torch_xla-{}-cp36-cp36m-linux_x86_64.whl'.format(CONFIG.wheels)
|
||||
TORCHVISION_WHEEL = 'torchvision-{}-cp36-cp36m-linux_x86_64.whl'.format(CONFIG.wheels)
|
||||
|
||||
# Update TPU XRT version
|
||||
def update_server_xrt():
|
||||
print('Updating server-side XRT to {} ...'.format(CONFIG.server))
|
||||
url = 'http://{TPU_ADDRESS}:8475/requestversion/{XRT_VERSION}'.format(
|
||||
TPU_ADDRESS=os.environ['COLAB_TPU_ADDR'].split(':')[0],
|
||||
XRT_VERSION=CONFIG.server,
|
||||
)
|
||||
print('Done updating server-side XRT: {}'.format(requests.post(url)))
|
||||
|
||||
update = threading.Thread(target=update_server_xrt)
|
||||
update.start()
|
||||
|
||||
.. code-block::
|
||||
|
||||
# Install Colab TPU compat PyTorch/TPU wheels and dependencies
|
||||
!pip uninstall -y torch torchvision
|
||||
!gsutil cp "$DIST_BUCKET/$TORCH_WHEEL" .
|
||||
!gsutil cp "$DIST_BUCKET/$TORCH_XLA_WHEEL" .
|
||||
!gsutil cp "$DIST_BUCKET/$TORCHVISION_WHEEL" .
|
||||
!pip install "$TORCH_WHEEL"
|
||||
!pip install "$TORCH_XLA_WHEEL"
|
||||
!pip install "$TORCHVISION_WHEEL"
|
||||
!sudo apt-get install libomp5
|
||||
update.join()
|
||||
|
||||
5. Once the above is done, install PyTorch Lightning (v 0.7.0+).
|
||||
|
||||
.. code-block::
|
||||
|
||||
!pip install pytorch-lightning
|
||||
|
||||
6. Then set up your LightningModule as normal.
|
||||
|
||||
---------------
|
||||
|
||||
DistributedSamplers
|
||||
-------------------
|
||||
Lightning automatically inserts the correct samplers - no need to do this yourself!
|
||||
|
||||
Usually, with TPUs (and DDP), you would need to define a DistributedSampler to move the right
|
||||
chunk of data to the appropriate TPU. As mentioned, this is not needed in Lightning
|
||||
|
||||
.. note:: Don't add distributedSamplers. Lightning does this automatically
|
||||
|
||||
If for some reason you still need to, this is how to construct the sampler
|
||||
for TPU use
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import torch_xla.core.xla_model as xm
|
||||
|
||||
def train_dataloader(self):
|
||||
dataset = MNIST(
|
||||
os.getcwd(),
|
||||
train=True,
|
||||
download=True,
|
||||
transform=transforms.ToTensor()
|
||||
)
|
||||
|
||||
# required for TPU support
|
||||
sampler = None
|
||||
if use_tpu:
|
||||
sampler = torch.utils.data.distributed.DistributedSampler(
|
||||
dataset,
|
||||
num_replicas=xm.xrt_world_size(),
|
||||
rank=xm.get_ordinal(),
|
||||
shuffle=True
|
||||
)
|
||||
|
||||
loader = DataLoader(
|
||||
dataset,
|
||||
sampler=sampler,
|
||||
batch_size=32
|
||||
)
|
||||
|
||||
return loader
|
||||
|
||||
Configure the number of TPU cores in the trainer. You can only choose 1 or 8.
|
||||
To use a full TPU pod skip to the TPU pod section.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import pytorch_lightning as pl
|
||||
|
||||
my_model = MyLightningModule()
|
||||
trainer = pl.Trainer(num_tpu_cores=8)
|
||||
trainer.fit(my_model)
|
||||
|
||||
That's it! Your model will train on all 8 TPU cores.
|
||||
|
||||
---------------
|
||||
|
||||
Distributed Backend with TPU
|
||||
----------------------------
|
||||
The ```distributed_backend``` option used for GPUs does not apply to TPUs.
|
||||
TPUs work in DDP mode by default (distributing over each core)
|
||||
|
||||
---------------
|
||||
|
||||
TPU Pod
|
||||
--------
|
||||
To train on more than 8 cores, your code actually doesn't change!
|
||||
All you need to do is submit the following command:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ python -m torch_xla.distributed.xla_dist
|
||||
--tpu=$TPU_POD_NAME
|
||||
--conda-env=torch-xla-nightly
|
||||
-- python /usr/share/torch-xla-0.5/pytorch/xla/test/test_train_imagenet.py --fake_data
|
||||
|
||||
---------------
|
||||
|
||||
16 bit precision
|
||||
-----------------
|
||||
Lightning also supports training in 16-bit precision with TPUs.
|
||||
By default, TPU training will use 32-bit precision. To enable 16-bit, also
|
||||
set the 16-bit flag.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import pytorch_lightning as pl
|
||||
|
||||
my_model = MyLightningModule()
|
||||
trainer = pl.Trainer(num_tpu_cores=8, precision=16)
|
||||
trainer.fit(my_model)
|
||||
|
||||
Under the hood the xla library will use the `bfloat16 type <https://en.wikipedia.org/wiki/Bfloat16_floating-point_format>`_.
|
||||
|
||||
---------------
|
||||
|
||||
About XLA
|
||||
----------
|
||||
XLA is the library that interfaces PyTorch with the TPUs.
|
||||
For more information check out `XLA <https://github.com/pytorch/xla>`_.
|
||||
|
||||
Guide for `troubleshooting XLA <https://github.com/pytorch/xla/blob/master/TROUBLESHOOTING.md>`_
|
||||
@@ -0,0 +1,24 @@
|
||||
.. role:: hidden
|
||||
:class: hidden-section
|
||||
|
||||
Trainer
|
||||
=======
|
||||
.. automodule:: pytorch_lightning.trainer
|
||||
:members: fit, test
|
||||
:noindex:
|
||||
:exclude-members:
|
||||
run_pretrain_routine,
|
||||
_abc_impl,
|
||||
_Trainer__set_random_port,
|
||||
_Trainer__set_root_gpu,
|
||||
_Trainer__init_optimizers,
|
||||
_Trainer__parse_gpu_ids,
|
||||
_Trainer__configure_schedulers,
|
||||
data_parallel,
|
||||
num_gpus,
|
||||
slurm_job_id,
|
||||
tng_tqdm_dic,
|
||||
training_tqdm_dict,
|
||||
progress_bar_dict,
|
||||
init_optimizers,
|
||||
configure_schedulers
|
||||
@@ -0,0 +1,111 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
|
||||
|
||||
Training Tricks
|
||||
================
|
||||
Lightning implements various tricks to help during training
|
||||
|
||||
Accumulate gradients
|
||||
-------------------------------------
|
||||
Accumulated gradients runs K small batches of size N before doing a backwards pass.
|
||||
The effect is a large effective batch size of size KxN.
|
||||
|
||||
.. seealso:: :class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT (ie: no accumulated grads)
|
||||
trainer = Trainer(accumulate_grad_batches=1)
|
||||
|
||||
|
||||
Gradient Clipping
|
||||
-------------------------------------
|
||||
Gradient clipping may be enabled to avoid exploding gradients. Specifically, this will `clip the gradient
|
||||
norm <https://pytorch.org/docs/stable/nn.html#torch.nn.utils.clip_grad_norm_>`_ computed over all model parameters together.
|
||||
|
||||
.. seealso:: :class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
|
||||
.. testcode::
|
||||
|
||||
# DEFAULT (ie: don't clip)
|
||||
trainer = Trainer(gradient_clip_val=0)
|
||||
|
||||
# clip gradients with norm above 0.5
|
||||
trainer = Trainer(gradient_clip_val=0.5)
|
||||
|
||||
Auto scaling of batch size
|
||||
--------------------------
|
||||
Auto scaling of batch size may be enabled to find the largest batch size that fits into
|
||||
memory. Larger batch size often yields better estimates of gradients, but may also result in
|
||||
longer training time.
|
||||
|
||||
.. seealso:: :class:`~pytorch_lightning.trainer.trainer.Trainer`
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# DEFAULT (ie: don't scale batch size automatically)
|
||||
trainer = Trainer(auto_scale_batch_size=None)
|
||||
|
||||
# Autoscale batch size
|
||||
trainer = Trainer(auto_scale_batch_size=None|'power'|'binsearch')
|
||||
|
||||
Currently, this feature supports two modes `'power'` scaling and `'binsearch'`
|
||||
scaling. In `'power'` scaling, starting from a batch size of 1 keeps doubling
|
||||
the batch size until an out-of-memory (OOM) error is encountered. Setting the
|
||||
argument to `'binsearch'` continues to finetune the batch size by performing
|
||||
a binary search.
|
||||
|
||||
.. note::
|
||||
|
||||
This feature expects that a `batch_size` field in the `hparams` of your model, i.e.,
|
||||
`model.hparams.batch_size` should exist and will be overridden by the results of this
|
||||
algorithm. Additionally, your `train_dataloader()` method should depend on this field
|
||||
for this feature to work i.e.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
def train_dataloader(self):
|
||||
return DataLoader(train_dataset, batch_size=self.hparams.batch_size)
|
||||
|
||||
.. warning::
|
||||
|
||||
Due to these constraints, this features does *NOT* work when passing dataloaders directly
|
||||
to `.fit()`.
|
||||
|
||||
The scaling algorithm has a number of parameters that the user can control by
|
||||
invoking the trainer method `.scale_batch_size` themself (see description below).
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
# Use default in trainer construction
|
||||
trainer = Trainer()
|
||||
|
||||
# Invoke method
|
||||
new_batch_size = trainer.scale_batch_size(model, ...)
|
||||
|
||||
# Override old batch size
|
||||
model.hparams.batch_size = new_batch_size
|
||||
|
||||
# Fit as normal
|
||||
trainer.fit(model)
|
||||
|
||||
The algorithm in short works by:
|
||||
1. Dumping the current state of the model and trainer
|
||||
2. Iteratively until convergence or maximum number of tries `max_trials` (default 25) has been reached:
|
||||
- Call `fit()` method of trainer. This evaluates `steps_per_trial` (default 3) number of
|
||||
training steps. Each training step can trigger an OOM error if the tensors
|
||||
(training batch, weights, gradients ect.) allocated during the steps have a
|
||||
too large memory footprint.
|
||||
- If an OOM error is encountered, decrease batch size else increase it.
|
||||
How much the batch size is increased/decreased is determined by the choosen
|
||||
stratrgy.
|
||||
3. The found batch size is saved to `model.hparams.batch_size`
|
||||
4. Restore the initial state of model and trainer
|
||||
|
||||
.. autoclass:: pytorch_lightning.trainer.training_tricks.TrainerTrainingTricksMixin
|
||||
:members: scale_batch_size
|
||||
:noindex:
|
||||
|
||||
.. warning:: Batch size finder is not supported for DDP yet, it is coming soon.
|
||||
@@ -0,0 +1,118 @@
|
||||
.. testsetup:: *
|
||||
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
Transfer Learning
|
||||
-----------------
|
||||
|
||||
Using Pretrained Models
|
||||
^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Sometimes we want to use a LightningModule as a pretrained model. This is fine because
|
||||
a LightningModule is just a `torch.nn.Module`!
|
||||
|
||||
.. note:: Remember that a LightningModule is EXACTLY a torch.nn.Module but with more capabilities.
|
||||
|
||||
Let's use the `AutoEncoder` as a feature extractor in a separate model.
|
||||
|
||||
|
||||
.. testcode::
|
||||
|
||||
class Encoder(torch.nn.Module):
|
||||
...
|
||||
|
||||
class AutoEncoder(LightningModule):
|
||||
def __init__(self):
|
||||
self.encoder = Encoder()
|
||||
self.decoder = Decoder()
|
||||
|
||||
class CIFAR10Classifier(LightningModule):
|
||||
def __init__(self):
|
||||
# init the pretrained LightningModule
|
||||
self.feature_extractor = AutoEncoder.load_from_checkpoint(PATH)
|
||||
self.feature_extractor.freeze()
|
||||
|
||||
# the autoencoder outputs a 100-dim representation and CIFAR-10 has 10 classes
|
||||
self.classifier = nn.Linear(100, 10)
|
||||
|
||||
def forward(self, x):
|
||||
representations = self.feature_extractor(x)
|
||||
x = self.classifier(representations)
|
||||
...
|
||||
|
||||
We used our pretrained Autoencoder (a LightningModule) for transfer learning!
|
||||
|
||||
Example: Imagenet (computer Vision)
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
.. testcode::
|
||||
:skipif: not TORCHVISION_AVAILABLE
|
||||
|
||||
import torchvision.models as models
|
||||
|
||||
class ImagenetTransferLearning(LightningModule):
|
||||
def __init__(self):
|
||||
# init a pretrained resnet
|
||||
num_target_classes = 10
|
||||
self.feature_extractor = models.resnet50(
|
||||
pretrained=True,
|
||||
num_classes=num_target_classes)
|
||||
self.feature_extractor.eval()
|
||||
|
||||
# use the pretrained model to classify cifar-10 (10 image classes)
|
||||
self.classifier = nn.Linear(2048, num_target_classes)
|
||||
|
||||
def forward(self, x):
|
||||
representations = self.feature_extractor(x)
|
||||
x = self.classifier(representations)
|
||||
...
|
||||
|
||||
Finetune
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = ImagenetTransferLearning()
|
||||
trainer = Trainer()
|
||||
trainer.fit(model)
|
||||
|
||||
And use it to predict your data of interest
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = ImagenetTransferLearning.load_from_checkpoint(PATH)
|
||||
model.freeze()
|
||||
|
||||
x = some_images_from_cifar10()
|
||||
predictions = model(x)
|
||||
|
||||
We used a pretrained model on imagenet, finetuned on CIFAR-10 to predict on CIFAR-10.
|
||||
In the non-academic world we would finetune on a tiny dataset you have and predict on your dataset.
|
||||
|
||||
Example: BERT (NLP)
|
||||
^^^^^^^^^^^^^^^^^^^
|
||||
Lightning is completely agnostic to what's used for transfer learning so long
|
||||
as it is a `torch.nn.Module` subclass.
|
||||
|
||||
Here's a model that uses `Huggingface transformers <https://github.com/huggingface/transformers>`_.
|
||||
|
||||
.. testcode::
|
||||
|
||||
class BertMNLIFinetuner(LightningModule):
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
|
||||
self.bert = BertModel.from_pretrained('bert-base-cased', output_attentions=True)
|
||||
self.W = nn.Linear(bert.config.hidden_size, 3)
|
||||
self.num_classes = 3
|
||||
|
||||
|
||||
def forward(self, input_ids, attention_mask, token_type_ids):
|
||||
|
||||
h, _, attn = self.bert(input_ids=input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids)
|
||||
|
||||
h_cls = h[:, 0]
|
||||
logits = self.W(h_cls)
|
||||
return logits, attn
|
||||
@@ -0,0 +1,141 @@
|
||||
.. testsetup:: *
|
||||
|
||||
import os
|
||||
from pytorch_lightning.trainer.trainer import Trainer
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
|
||||
Saving and loading weights
|
||||
==========================
|
||||
|
||||
Lightning can automate saving and loading checkpoints.
|
||||
|
||||
Checkpoint saving
|
||||
-----------------
|
||||
A Lightning checkpoint has everything needed to restore a training session including:
|
||||
|
||||
- 16-bit scaling factor (apex)
|
||||
- Current epoch
|
||||
- Global step
|
||||
- Model state_dict
|
||||
- State of all optimizers
|
||||
- State of all learningRate schedulers
|
||||
- State of all callbacks
|
||||
- The hyperparameters used for that model if passed in as hparams (Argparse.Namespace)
|
||||
|
||||
Automatic saving
|
||||
^^^^^^^^^^^^^^^^
|
||||
|
||||
Checkpointing is enabled by default to the current working directory.
|
||||
To change the checkpoint path pass in:
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(default_save_path='/your/path/to/save/checkpoints')
|
||||
|
||||
To modify the behavior of checkpointing pass in your own callback.
|
||||
|
||||
.. testcode::
|
||||
|
||||
from pytorch_lightning.callbacks import ModelCheckpoint
|
||||
|
||||
# DEFAULTS used by the Trainer
|
||||
checkpoint_callback = ModelCheckpoint(
|
||||
filepath=os.getcwd(),
|
||||
save_top_k=True,
|
||||
verbose=True,
|
||||
monitor='val_loss',
|
||||
mode='min',
|
||||
prefix=''
|
||||
)
|
||||
|
||||
trainer = Trainer(checkpoint_callback=checkpoint_callback)
|
||||
|
||||
|
||||
Or disable it by passing
|
||||
|
||||
.. testcode::
|
||||
|
||||
trainer = Trainer(checkpoint_callback=False)
|
||||
|
||||
|
||||
The Lightning checkpoint also saves the hparams (hyperparams) passed into the LightningModule init.
|
||||
|
||||
.. note:: hparams is a `Namespace <https://docs.python.org/2/library/argparse.html#argparse.Namespace>`_.
|
||||
|
||||
.. testcode::
|
||||
|
||||
from argparse import Namespace
|
||||
|
||||
# usually these come from command line args
|
||||
args = Namespace(learning_rate=0.001)
|
||||
|
||||
# define you module to have hparams as the first arg
|
||||
# this means your checkpoint will have everything that went into making
|
||||
# this model (in this case, learning rate)
|
||||
class MyLightningModule(LightningModule):
|
||||
|
||||
def __init__(self, hparams, *args, **kwargs):
|
||||
self.hparams = hparams
|
||||
|
||||
Manual saving
|
||||
^^^^^^^^^^^^^
|
||||
You can manually save checkpoints and restore your model from the checkpointed state.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = MyLightningModule(hparams)
|
||||
trainer.fit(model)
|
||||
trainer.save_checkpoint("example.ckpt")
|
||||
new_model = MyModel.load_from_checkpoint(checkpoint_path="example.ckpt")
|
||||
|
||||
Checkpoint Loading
|
||||
------------------
|
||||
|
||||
To load a model along with its weights, biases and hyperparameters use following method.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = MyLightingModule.load_from_checkpoint(PATH)
|
||||
model.eval()
|
||||
y_hat = model(x)
|
||||
|
||||
The above only works if you used `hparams` in your model definition
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def __init__(self, hparams):
|
||||
self.hparams = hparams
|
||||
self.l1 = nn.Linear(hparams.in_dim, hparams.out_dim)
|
||||
|
||||
But if you don't and instead pass individual parameters
|
||||
|
||||
.. testcode::
|
||||
|
||||
class LitModel(LightningModule):
|
||||
|
||||
def __init__(self, in_dim, out_dim):
|
||||
self.l1 = nn.Linear(in_dim, out_dim)
|
||||
|
||||
you can restore the model like this
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitModel.load_from_checkpoint(PATH, in_dim=128, out_dim=10)
|
||||
|
||||
|
||||
Restoring Training State
|
||||
------------------------
|
||||
|
||||
If you don't just want to load weights, but instead restore the full training,
|
||||
do the following:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
model = LitModel()
|
||||
trainer = Trainer(resume_from_checkpoint='some/path/to/my_checkpoint.ckpt')
|
||||
|
||||
# automatically restores model, epoch, step, LR schedulers, apex, etc...
|
||||
trainer.fit(model)
|
||||
@@ -0,0 +1,36 @@
|
||||
# This is Conda environment file
|
||||
# Usage: `conda env update -f environment.yml`
|
||||
|
||||
channels:
|
||||
- conda-forge
|
||||
- pytorch
|
||||
|
||||
dependencies:
|
||||
- python==3.7.6
|
||||
- pip==20.0.2
|
||||
- tqdm>=4.35.0
|
||||
- numpy>=1.16.4
|
||||
- pytorch>=1.1
|
||||
- tensorboard>=1.14
|
||||
- future>=0.17.1
|
||||
- pyyaml>=3.13
|
||||
|
||||
# For dev and testing
|
||||
- tox
|
||||
- coverage
|
||||
- codecov
|
||||
- pytest>=3.0.5
|
||||
- pytest-cov
|
||||
- pytest-flake8
|
||||
- flake8
|
||||
- autopep8
|
||||
- check-manifest
|
||||
- twine==1.13.0
|
||||
|
||||
- pip:
|
||||
- test-tube>=0.7.5
|
||||
- mlflow>=1.0.0
|
||||
- comet_ml>=1.0.56
|
||||
- wandb>=0.8.21
|
||||
- neptune-client>=0.4.4
|
||||
- trains>=0.13.3
|
||||
@@ -1,11 +1,67 @@
|
||||
# Examples
|
||||
This folder has 3 sections:
|
||||
|
||||
### Domain templates
|
||||
These are templates to show common approaches such as GANs and RL.
|
||||
## Basic Examples
|
||||
Use these examples to test how lightning works.
|
||||
|
||||
### Basic examples
|
||||
These show the most common use of Lightning for either CPU or GPU training.
|
||||
#### Test on CPU
|
||||
```bash
|
||||
python cpu_template.py
|
||||
```
|
||||
|
||||
### Multi-node examples
|
||||
These show how to run jobs on a GPU cluster using lightning.
|
||||
---
|
||||
#### Train on a single GPU
|
||||
```bash
|
||||
python gpu_template.py --gpus 1
|
||||
```
|
||||
|
||||
---
|
||||
#### DataParallel (dp)
|
||||
Train on multiple GPUs using DataParallel.
|
||||
|
||||
```bash
|
||||
python gpu_template.py --gpus 2 --distributed_backend dp
|
||||
```
|
||||
|
||||
---
|
||||
#### DistributedDataParallel (ddp)
|
||||
|
||||
Train on multiple GPUs using DistributedDataParallel
|
||||
```bash
|
||||
python gpu_template.py --gpus 2 --distributed_backend ddp
|
||||
```
|
||||
|
||||
---
|
||||
#### DistributedDataParallel+DP (ddp2)
|
||||
|
||||
Train on multiple GPUs using DistributedDataParallel + dataparallel.
|
||||
On a single node, uses all GPUs for 1 model. Then shares gradient information
|
||||
across nodes.
|
||||
```bash
|
||||
python gpu_template.py --gpus 2 --distributed_backend ddp2
|
||||
```
|
||||
|
||||
## Multi-node example
|
||||
|
||||
This demo launches a job using 2 GPUs on 2 different nodes (4 GPUs total).
|
||||
To run this demo do the following:
|
||||
|
||||
1. Log into the jumphost node of your SLURM-managed cluster.
|
||||
2. Create a conda environment with Lightning and a GPU PyTorch version.
|
||||
3. Choose a script to submit
|
||||
|
||||
### DDP
|
||||
Submit this job to run with DistributedDataParallel (2 nodes, 2 gpus each)
|
||||
```bash
|
||||
sbatch ddp_job_submit.sh YourEnv
|
||||
```
|
||||
|
||||
### DDP2
|
||||
Submit this job to run with a different implementation of DistributedDataParallel.
|
||||
In this version, each node acts like DataParallel but syncs across nodes like DDP.
|
||||
```bash
|
||||
sbatch ddp2_job_submit.sh YourEnv
|
||||
```
|
||||
|
||||
## Domain templates
|
||||
These are templates to show common approaches such as GANs and RL.
|
||||
|
||||
@@ -3,13 +3,13 @@ Template model definition
|
||||
-------------------------
|
||||
|
||||
In 99% of cases you want to just copy `one of the examples
|
||||
<https://github.com/williamFalcon/pytorch-lightning/tree/master/pl_examples>`_
|
||||
to start a new lightningModule and change the core of what your model is actually trying to do.
|
||||
<https://github.com/PyTorchLightning/pytorch-lightning/tree/master/pl_examples>`_
|
||||
to start a new lightningModule and change the core of what your model is actually trying to do.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# get a copy of the module template
|
||||
wget https://raw.githubusercontent.com/williamFalcon/pytorch-lightning/master/pl_examples/new_project_templates/lightning_module_template.py # noqa: E501
|
||||
wget https://raw.githubusercontent.com/PyTorchLightning/pytorch-lightning/master/pl_examples/new_project_templates/lightning_module_template.py # noqa: E501
|
||||
|
||||
|
||||
Trainer Example
|
||||
@@ -43,6 +43,7 @@ Normally, we want to let the `__main__` function start the training.
|
||||
|
||||
The main function is your entry into the program. This is where you init your model, checkpoint directory,
|
||||
and launch the training. The main function should have 3 arguments:
|
||||
|
||||
- hparams: a configuration of hyperparameters.
|
||||
- slurm_manager: Slurm cluster manager object (can be None)
|
||||
- dict: for you to return any values you want (useful in meta-learning, otherwise set to)
|
||||
@@ -139,7 +140,7 @@ Hyperparameter search on a SLURM HPC cluster
|
||||
|
||||
"""
|
||||
|
||||
from .basic_examples.lightning_module_template import LightningTemplateModel
|
||||
from pl_examples.models.lightning_template import LightningTemplateModel
|
||||
|
||||
__all__ = [
|
||||
'LightningTemplateModel'
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Basic Examples
|
||||
## Basic Examples
|
||||
Use these examples to test how lightning works.
|
||||
|
||||
#### Test on CPU
|
||||
@@ -31,9 +31,32 @@ python gpu_template.py --gpus 2 --distributed_backend ddp
|
||||
---
|
||||
#### DistributedDataParallel+DP (ddp2)
|
||||
|
||||
Train on multiple GPUs using DistributedDataParallel + dataparallel.
|
||||
Train on multiple GPUs using DistributedDataParallel + DataParallel.
|
||||
On a single node, uses all GPUs for 1 model. Then shares gradient information
|
||||
across nodes.
|
||||
```bash
|
||||
python gpu_template.py --gpus 2 --distributed_backend ddp2
|
||||
```
|
||||
```
|
||||
|
||||
|
||||
# Multi-node example
|
||||
|
||||
This demo launches a job using 2 GPUs on 2 different nodes (4 GPUs total).
|
||||
To run this demo do the following:
|
||||
|
||||
1. Log into the jumphost node of your SLURM-managed cluster.
|
||||
2. Create a conda environment with Lightning and a GPU PyTorch version.
|
||||
3. Choose a script to submit
|
||||
|
||||
#### DDP
|
||||
Submit this job to run with DistributedDataParallel (2 nodes, 2 gpus each)
|
||||
```bash
|
||||
sbatch ddp_job_submit.sh YourEnv
|
||||
```
|
||||
|
||||
#### DDP2
|
||||
Submit this job to run with a different implementation of DistributedDataParallel.
|
||||
In this version, each node acts like DataParallel but syncs across nodes like DDP.
|
||||
```bash
|
||||
sbatch ddp2_job_submit.sh YourEnv
|
||||
```
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
"""
|
||||
Runs a model on a single node across N-gpus.
|
||||
Runs a model on the CPU on a single node.
|
||||
"""
|
||||
import os
|
||||
from argparse import ArgumentParser
|
||||
@@ -7,8 +7,8 @@ from argparse import ArgumentParser
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from pl_examples.basic_examples.lightning_module_template import LightningTemplateModel
|
||||
from pytorch_lightning import Trainer
|
||||
import pytorch_lightning as pl
|
||||
from pl_examples.models.lightning_template import LightningTemplateModel
|
||||
|
||||
SEED = 2334
|
||||
torch.manual_seed(SEED)
|
||||
@@ -28,7 +28,7 @@ def main(hparams):
|
||||
# ------------------------
|
||||
# 2 INIT TRAINER
|
||||
# ------------------------
|
||||
trainer = Trainer()
|
||||
trainer = pl.Trainer(max_epochs=hparams.epochs, overfit_pct=0.01, early_stop_callback=True)
|
||||
|
||||
# ------------------------
|
||||
# 3 START TRAINING
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
"""
|
||||
Runs a model on a single node across N-gpus.
|
||||
Runs a model on a single node across multiple gpus.
|
||||
"""
|
||||
import os
|
||||
from argparse import ArgumentParser
|
||||
@@ -7,8 +7,8 @@ from argparse import ArgumentParser
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from pl_examples.basic_examples.lightning_module_template import LightningTemplateModel
|
||||
from pytorch_lightning import Trainer
|
||||
import pytorch_lightning as pl
|
||||
from pl_examples.models.lightning_template import LightningTemplateModel
|
||||
|
||||
SEED = 2334
|
||||
torch.manual_seed(SEED)
|
||||
@@ -28,10 +28,11 @@ def main(hparams):
|
||||
# ------------------------
|
||||
# 2 INIT TRAINER
|
||||
# ------------------------
|
||||
trainer = Trainer(
|
||||
trainer = pl.Trainer(
|
||||
max_epochs=hparams.epochs,
|
||||
gpus=hparams.gpus,
|
||||
distributed_backend=hparams.distributed_backend,
|
||||
use_amp=hparams.use_16bit
|
||||
precision=16 if hparams.use_16bit else 32,
|
||||
)
|
||||
|
||||
# ------------------------
|
||||
|
||||
@@ -1,258 +0,0 @@
|
||||
"""
|
||||
Example template for defining a system
|
||||
"""
|
||||
import os
|
||||
import logging
|
||||
from argparse import ArgumentParser
|
||||
from collections import OrderedDict
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
import torchvision.transforms as transforms
|
||||
from torch import optim
|
||||
from torch.utils.data import DataLoader
|
||||
from torch.utils.data.distributed import DistributedSampler
|
||||
from torchvision.datasets import MNIST
|
||||
|
||||
import pytorch_lightning as pl
|
||||
from pytorch_lightning.core.lightning import LightningModule
|
||||
|
||||
|
||||
class LightningTemplateModel(LightningModule):
|
||||
"""
|
||||
Sample model to show how to define a template
|
||||
"""
|
||||
|
||||
def __init__(self, hparams):
|
||||
"""
|
||||
Pass in parsed HyperOptArgumentParser to the model
|
||||
:param hparams:
|
||||
"""
|
||||
# init superclass
|
||||
super(LightningTemplateModel, self).__init__()
|
||||
self.hparams = hparams
|
||||
|
||||
self.batch_size = hparams.batch_size
|
||||
|
||||
# if you specify an example input, the summary will show input/output for each layer
|
||||
self.example_input_array = torch.rand(5, 28 * 28)
|
||||
|
||||
# build model
|
||||
self.__build_model()
|
||||
|
||||
# ---------------------
|
||||
# MODEL SETUP
|
||||
# ---------------------
|
||||
def __build_model(self):
|
||||
"""
|
||||
Layout model
|
||||
:return:
|
||||
"""
|
||||
self.c_d1 = nn.Linear(in_features=self.hparams.in_features,
|
||||
out_features=self.hparams.hidden_dim)
|
||||
self.c_d1_bn = nn.BatchNorm1d(self.hparams.hidden_dim)
|
||||
self.c_d1_drop = nn.Dropout(self.hparams.drop_prob)
|
||||
|
||||
self.c_d2 = nn.Linear(in_features=self.hparams.hidden_dim,
|
||||
out_features=self.hparams.out_features)
|
||||
|
||||
# ---------------------
|
||||
# TRAINING
|
||||
# ---------------------
|
||||
def forward(self, x):
|
||||
"""
|
||||
No special modification required for lightning, define as you normally would
|
||||
:param x:
|
||||
:return:
|
||||
"""
|
||||
|
||||
x = self.c_d1(x)
|
||||
x = torch.tanh(x)
|
||||
x = self.c_d1_bn(x)
|
||||
x = self.c_d1_drop(x)
|
||||
|
||||
x = self.c_d2(x)
|
||||
logits = F.log_softmax(x, dim=1)
|
||||
|
||||
return logits
|
||||
|
||||
def loss(self, labels, logits):
|
||||
nll = F.nll_loss(logits, labels)
|
||||
return nll
|
||||
|
||||
def training_step(self, batch, batch_idx):
|
||||
"""
|
||||
Lightning calls this inside the training loop
|
||||
:param batch:
|
||||
:return:
|
||||
"""
|
||||
# forward pass
|
||||
x, y = batch
|
||||
x = x.view(x.size(0), -1)
|
||||
|
||||
y_hat = self.forward(x)
|
||||
|
||||
# calculate loss
|
||||
loss_val = self.loss(y, y_hat)
|
||||
|
||||
# in DP mode (default) make sure if result is scalar, there's another dim in the beginning
|
||||
if self.trainer.use_dp or self.trainer.use_ddp2:
|
||||
loss_val = loss_val.unsqueeze(0)
|
||||
|
||||
tqdm_dict = {'train_loss': loss_val}
|
||||
output = OrderedDict({
|
||||
'loss': loss_val,
|
||||
'progress_bar': tqdm_dict,
|
||||
'log': tqdm_dict
|
||||
})
|
||||
|
||||
# can also return just a scalar instead of a dict (return loss_val)
|
||||
return output
|
||||
|
||||
def validation_step(self, batch, batch_idx):
|
||||
"""
|
||||
Lightning calls this inside the validation loop
|
||||
:param batch:
|
||||
:return:
|
||||
"""
|
||||
x, y = batch
|
||||
x = x.view(x.size(0), -1)
|
||||
y_hat = self.forward(x)
|
||||
|
||||
loss_val = self.loss(y, y_hat)
|
||||
|
||||
# acc
|
||||
labels_hat = torch.argmax(y_hat, dim=1)
|
||||
val_acc = torch.sum(y == labels_hat).item() / (len(y) * 1.0)
|
||||
val_acc = torch.tensor(val_acc)
|
||||
|
||||
if self.on_gpu:
|
||||
val_acc = val_acc.cuda(loss_val.device.index)
|
||||
|
||||
# in DP mode (default) make sure if result is scalar, there's another dim in the beginning
|
||||
if self.trainer.use_dp or self.trainer.use_ddp2:
|
||||
loss_val = loss_val.unsqueeze(0)
|
||||
val_acc = val_acc.unsqueeze(0)
|
||||
|
||||
output = OrderedDict({
|
||||
'val_loss': loss_val,
|
||||
'val_acc': val_acc,
|
||||
})
|
||||
|
||||
# can also return just a scalar instead of a dict (return loss_val)
|
||||
return output
|
||||
|
||||
def validation_end(self, outputs):
|
||||
"""
|
||||
Called at the end of validation to aggregate outputs
|
||||
:param outputs: list of individual outputs of each validation step
|
||||
:return:
|
||||
"""
|
||||
# if returned a scalar from validation_step, outputs is a list of tensor scalars
|
||||
# we return just the average in this case (if we want)
|
||||
# return torch.stack(outputs).mean()
|
||||
|
||||
val_loss_mean = 0
|
||||
val_acc_mean = 0
|
||||
for output in outputs:
|
||||
val_loss = output['val_loss']
|
||||
|
||||
# reduce manually when using dp
|
||||
if self.trainer.use_dp or self.trainer.use_ddp2:
|
||||
val_loss = torch.mean(val_loss)
|
||||
val_loss_mean += val_loss
|
||||
|
||||
# reduce manually when using dp
|
||||
val_acc = output['val_acc']
|
||||
if self.trainer.use_dp or self.trainer.use_ddp2:
|
||||
val_acc = torch.mean(val_acc)
|
||||
|
||||
val_acc_mean += val_acc
|
||||
|
||||
val_loss_mean /= len(outputs)
|
||||
val_acc_mean /= len(outputs)
|
||||
tqdm_dict = {'val_loss': val_loss_mean, 'val_acc': val_acc_mean}
|
||||
result = {'progress_bar': tqdm_dict, 'log': tqdm_dict, 'val_loss': val_loss_mean}
|
||||
return result
|
||||
|
||||
# ---------------------
|
||||
# TRAINING SETUP
|
||||
# ---------------------
|
||||
def configure_optimizers(self):
|
||||
"""
|
||||
return whatever optimizers we want here
|
||||
:return: list of optimizers
|
||||
"""
|
||||
optimizer = optim.Adam(self.parameters(), lr=self.hparams.learning_rate)
|
||||
scheduler = optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=10)
|
||||
return [optimizer], [scheduler]
|
||||
|
||||
def __dataloader(self, train):
|
||||
# init data generators
|
||||
transform = transforms.Compose([transforms.ToTensor(),
|
||||
transforms.Normalize((0.5,), (1.0,))])
|
||||
dataset = MNIST(root=self.hparams.data_root, train=train,
|
||||
transform=transform, download=True)
|
||||
|
||||
# when using multi-node (ddp) we need to add the datasampler
|
||||
train_sampler = None
|
||||
batch_size = self.hparams.batch_size
|
||||
|
||||
if self.use_ddp:
|
||||
train_sampler = DistributedSampler(dataset)
|
||||
|
||||
should_shuffle = train_sampler is None
|
||||
loader = DataLoader(
|
||||
dataset=dataset,
|
||||
batch_size=batch_size,
|
||||
shuffle=should_shuffle,
|
||||
sampler=train_sampler,
|
||||
num_workers=0
|
||||
)
|
||||
|
||||
return loader
|
||||
|
||||
@pl.data_loader
|
||||
def train_dataloader(self):
|
||||
logging.info('training data loader called')
|
||||
return self.__dataloader(train=True)
|
||||
|
||||
@pl.data_loader
|
||||
def val_dataloader(self):
|
||||
logging.info('val data loader called')
|
||||
return self.__dataloader(train=False)
|
||||
|
||||
@pl.data_loader
|
||||
def test_dataloader(self):
|
||||
logging.info('test data loader called')
|
||||
return self.__dataloader(train=False)
|
||||
|
||||
@staticmethod
|
||||
def add_model_specific_args(parent_parser, root_dir): # pragma: no cover
|
||||
"""
|
||||
Parameters you define here will be available to your model through self.hparams
|
||||
:param parent_parser:
|
||||
:param root_dir:
|
||||
:return:
|
||||
"""
|
||||
parser = ArgumentParser(parents=[parent_parser])
|
||||
|
||||
# param overwrites
|
||||
# parser.set_defaults(gradient_clip_val=5.0)
|
||||
|
||||
# network params
|
||||
parser.add_argument('--in_features', default=28 * 28, type=int)
|
||||
parser.add_argument('--out_features', default=10, type=int)
|
||||
# use 500 for CPU, 50000 for GPU to see speed difference
|
||||
parser.add_argument('--hidden_dim', default=50000, type=int)
|
||||
parser.add_argument('--drop_prob', default=0.2, type=float)
|
||||
parser.add_argument('--learning_rate', default=0.001, type=float)
|
||||
|
||||
# data
|
||||
parser.add_argument('--data_root', default=os.path.join(root_dir, 'mnist'), type=str)
|
||||
|
||||
# training params (opt)
|
||||
parser.add_argument('--optimizer_name', default='adam', type=str)
|
||||
parser.add_argument('--batch_size', default=64, type=int)
|
||||
return parser
|
||||