Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6132e76d90 | ||
|
|
4cb8cbf83e | ||
|
|
d5057da9bb | ||
|
|
24904a5999 | ||
|
|
881a2b45c5 | ||
|
|
41fc83ad03 | ||
|
|
9928276dc3 | ||
|
|
6607f93f5e | ||
|
|
dcc0cfe06e | ||
|
|
00d5c5c4bf | ||
|
|
f8152dde19 | ||
|
|
0586e04c21 | ||
|
|
3de18a7fab | ||
|
|
dc0bf24cc5 | ||
|
|
8ce4c3070c | ||
|
|
13c3acb976 | ||
|
|
c1150ff584 | ||
|
|
4f11d70f7e | ||
|
|
2bf9a2b317 | ||
|
|
b1e0ad0c4f | ||
|
|
5624f92f02 | ||
|
|
1335032954 | ||
|
|
0b13c66e07 | ||
|
|
ef25b54926 | ||
|
|
064dbfeefa | ||
|
|
044c69e7a5 | ||
|
|
32a46e7471 | ||
|
|
b52d59822d | ||
|
|
529995ecde | ||
|
|
0726328c92 | ||
|
|
090e83c286 | ||
|
|
c3802626b9 | ||
|
|
9570c2477f | ||
|
|
2b1a898b2a | ||
|
|
f2a42aa66e | ||
|
|
30a03f1fd1 | ||
|
|
ca25448f59 | ||
|
|
8f922710b0 | ||
|
|
9509c6ab9d | ||
|
|
ccba176979 | ||
|
|
1b8f383897 | ||
|
|
e5cd9e86d2 | ||
|
|
a7e86a4f26 | ||
|
|
8f0c73b32c | ||
|
|
c7d7b48a91 | ||
|
|
583eb90f07 | ||
|
|
c3a9249c0c | ||
|
|
81b70e779c | ||
|
|
085713a818 | ||
|
|
5af7c851dc | ||
|
|
691312d467 | ||
|
|
650c256c13 | ||
|
|
9b00c4380b | ||
|
|
4851457e93 | ||
|
|
1c54878fab | ||
|
|
c139b84454 | ||
|
|
15c38b4ca6 | ||
|
|
271c949a71 | ||
|
|
09b5401434 | ||
|
|
dad76547f0 | ||
|
|
c14a110c81 | ||
|
|
d1e16546cc | ||
|
|
f91bc3d0c9 | ||
|
|
5168808b6c | ||
|
|
501cca7b5e | ||
|
|
f19d40d858 | ||
|
|
5a721ce01d | ||
|
|
66421fb4d8 | ||
|
|
3658ee8c88 | ||
|
|
afaab4bb02 | ||
|
|
c1e004bdff | ||
|
|
874e5a3ef5 | ||
|
|
51529f370c | ||
|
|
5ddf98866c | ||
|
|
fbb6830876 | ||
|
|
ee1bb281da | ||
|
|
da2f7e88fe | ||
|
|
3fb28e353f | ||
|
|
c8e7f44f0a | ||
|
|
3d9049aeeb | ||
|
|
7b8726af24 | ||
|
|
0443b05360 | ||
|
|
e422dbdef2 | ||
|
|
ec5fc0b1c8 | ||
|
|
c52e3f20ba | ||
|
|
e367dceceb | ||
|
|
19b8666808 | ||
|
|
a2df0e9fca | ||
|
|
73b094550e | ||
|
|
d05ae109e1 | ||
|
|
fdb25791d5 | ||
|
|
5a88492498 | ||
|
|
f19a12b829 | ||
|
|
3ed0ea73f4 | ||
|
|
b98fd24a72 | ||
|
|
eb4e9a0f91 | ||
|
|
ec7c136b0a | ||
|
|
ee7e43cc1a | ||
|
|
ea9f5f3c5c | ||
|
|
bd44798412 | ||
|
|
2af0cc0f80 | ||
|
|
5c0fa14d2d | ||
|
|
7cc50d3203 | ||
|
|
9c7da13177 | ||
|
|
45e0645f0b | ||
|
|
cb884cc74a | ||
|
|
d3a6475580 | ||
|
|
4e7061b2db | ||
|
|
0250879f65 | ||
|
|
5cee23ae68 | ||
|
|
d518558b3d | ||
|
|
5c085c843f | ||
|
|
936b434ac4 | ||
|
|
43059c9fd9 | ||
|
|
19f72d426d | ||
|
|
14ecaf3023 | ||
|
|
80c93a9b6d | ||
|
|
06a1b4dc57 | ||
|
|
4e779eedf1 | ||
|
|
afb341f5fd | ||
|
|
e34b0fa115 | ||
|
|
6fa0345f25 | ||
|
|
dd43ae6639 | ||
|
|
840385e0a7 | ||
|
|
9daaf51a1f | ||
|
|
9b2511e54e | ||
|
|
b6674e6540 | ||
|
|
a109cb6d44 | ||
|
|
f730d6b9de | ||
|
|
0947f2792d | ||
|
|
dbf19acde4 | ||
|
|
1aaa833153 | ||
|
|
c73d995680 | ||
|
|
3fe5028724 | ||
|
|
44dd24f910 | ||
|
|
c20a9c4e63 | ||
|
|
edffdf34b7 | ||
|
|
340c24c5b4 | ||
|
|
11f20f3f08 | ||
|
|
199b330aa5 | ||
|
|
5143db1024 | ||
|
|
2f0bd3de15 | ||
|
|
4d6c541967 | ||
|
|
5228a5c978 | ||
|
|
c6118bd17d | ||
|
|
d2c9e84af4 | ||
|
|
fe0a0ff055 | ||
|
|
40e287245d | ||
|
|
c79646a537 | ||
|
|
e293e1c2fb | ||
|
|
f2ab65bd03 | ||
|
|
aa686516d3 | ||
|
|
11c80a0251 | ||
|
|
c5e9e5812b | ||
|
|
98b126fa0b | ||
|
|
55553f9e69 | ||
|
|
563013c745 | ||
|
|
fefb9778a7 | ||
|
|
be1831082c | ||
|
|
9e15ee2e8a | ||
|
|
ceff7d7271 | ||
|
|
df9b5cd6f0 | ||
|
|
bc57801525 | ||
|
|
1b403373d3 | ||
|
|
987fb74ca6 | ||
|
|
65a209bc7b | ||
|
|
16e6d9e90f | ||
|
|
7ce6bee763 | ||
|
|
6d7ca3eb55 | ||
|
|
8e9f205e9a | ||
|
|
5fdb6c4368 | ||
|
|
71ebfd402c | ||
|
|
6e4ca83531 | ||
|
|
05793d6a3c | ||
|
|
3dc374b7db | ||
|
|
3ac8a4f617 | ||
|
|
21b4b5b063 | ||
|
|
9ffd921ca4 | ||
|
|
14e6ebbb95 | ||
|
|
277b685a39 | ||
|
|
67e7715ed9 | ||
|
|
bb87900209 | ||
|
|
96ebe7286b | ||
|
|
0b3a5dde09 | ||
|
|
17a30360b4 | ||
|
|
b386f51916 | ||
|
|
dd4f40c7f4 | ||
|
|
d402fc085a | ||
|
|
a12b9cd60f | ||
|
|
bf1032d550 | ||
|
|
dea05e1e36 | ||
|
|
a9f0117c82 | ||
|
|
83f1fe9ecd | ||
|
|
7344274037 | ||
|
|
e36bfa9a3b | ||
|
|
950aa245b8 | ||
|
|
ac47b2e370 | ||
|
|
6662fc809b | ||
|
|
12b3171d5e | ||
|
|
579d4751bd | ||
|
|
d4c607323d | ||
|
|
3ad30738a7 | ||
|
|
119273ae7e | ||
|
|
95bc39a685 | ||
|
|
b054104851 | ||
|
|
e0e0cf849a | ||
|
|
c95b3d7088 | ||
|
|
19c2bdb008 | ||
|
|
3c815f3888 | ||
|
|
cfcd9b29fd | ||
|
|
4fca7d98b2 | ||
|
|
dea950bcb6 | ||
|
|
4c43755fb6 | ||
|
|
374942a9e9 | ||
|
|
8add418428 | ||
|
|
9d73cc4574 | ||
|
|
5edb4bb1bf | ||
|
|
0e811868d3 | ||
|
|
9efcfa25e8 | ||
|
|
48036cf581 | ||
|
|
79fd9b4049 | ||
|
|
46b2181a4e | ||
|
|
d04f25a8bd | ||
|
|
9f341c350c | ||
|
|
f22f97f680 | ||
|
|
fe75745d44 | ||
|
|
bbc9f1337e | ||
|
|
eb1fa9213a | ||
|
|
a7033a6527 | ||
|
|
7e6c2d69c8 | ||
|
|
1a21b81804 | ||
|
|
2bbd520613 | ||
|
|
8285e4ebd1 | ||
|
|
859849894e | ||
|
|
e74dd48ae0 | ||
|
|
a14eb71210 | ||
|
|
0b6a9718ff | ||
|
|
6ac0d8a029 | ||
|
|
2ee013bcab | ||
|
|
1ebb3f5714 | ||
|
|
11c5134961 | ||
|
|
e328b7268f | ||
|
|
d297e99e5a | ||
|
|
48c7a4c82e | ||
|
|
eb3b52f863 | ||
|
|
4e58cca127 | ||
|
|
182a98768f | ||
|
|
f483447235 | ||
|
|
c59050608b | ||
|
|
3b9844f92c | ||
|
|
ee301a22f6 | ||
|
|
b7d16f66aa | ||
|
|
57aad5b802 | ||
|
|
9e311433ba | ||
|
|
04b310927a | ||
|
|
2fed3ad014 | ||
|
|
0ec94e0af6 | ||
|
|
9470be0900 | ||
|
|
479a0c6271 | ||
|
|
053f9c397f | ||
|
|
2c82469756 | ||
|
|
fdfc7009d0 | ||
|
|
59fcfe137d | ||
|
|
9b434b32bc | ||
|
|
fc27cd8628 | ||
|
|
a62f03c396 | ||
|
|
a270439814 | ||
|
|
a97af4a078 | ||
|
|
d7127cc22f | ||
|
|
802ab4edd8 | ||
|
|
ae7f28fb31 | ||
|
|
f63e3e6e4c | ||
|
|
b9bb497ebc | ||
|
|
4368b9e7f8 | ||
|
|
2e9cb5d2cf | ||
|
|
9ec8a93e05 | ||
|
|
3ed8778656 | ||
|
|
bf5e3cf870 | ||
|
|
c23215893d | ||
|
|
1c96f7ca71 | ||
|
|
0d5661835b | ||
|
|
fe1a3c0bc9 | ||
|
|
82bffb87f1 | ||
|
|
741c6732f1 | ||
|
|
9d8caf7888 | ||
|
|
e63d354413 | ||
|
|
831c515477 | ||
|
|
06ba5d6804 | ||
|
|
0137cd106e | ||
|
|
8b3b63d714 | ||
|
|
bf79916f29 | ||
|
|
6bf462f79d | ||
|
|
106cdee495 | ||
|
|
dfb7301733 | ||
|
|
da707b2cbc | ||
|
|
873ba9dde9 | ||
|
|
15c452f39d | ||
|
|
be7111815b | ||
|
|
17db1a952b | ||
|
|
455143d0f0 | ||
|
|
f0892852cb | ||
|
|
ca48556d0c | ||
|
|
f961aa3174 | ||
|
|
64c8eca7df | ||
|
|
b3332c1742 | ||
|
|
b35f697a75 | ||
|
|
3bc32a1d48 | ||
|
|
d66f851554 | ||
|
|
d733f107e1 | ||
|
|
805e2e1c83 | ||
|
|
03dc17d9d7 | ||
|
|
a4909f823e | ||
|
|
e91b259595 | ||
|
|
14284046e4 | ||
|
|
59a9a5e6ba | ||
|
|
f3b0a9e0c0 | ||
|
|
92b572a364 | ||
|
|
014be9b530 | ||
|
|
fa675e0082 | ||
|
|
113cbb9709 | ||
|
|
b72bdc8112 | ||
|
|
0842fa8354 | ||
|
|
4e4f3f4095 | ||
|
|
d014febeb9 | ||
|
|
a3264df643 | ||
|
|
f0208e3e37 | ||
|
|
f84e1b9cbc | ||
|
|
16b2086031 | ||
|
|
0e24ba565d | ||
|
|
49718b02a5 | ||
|
|
63031ed364 | ||
|
|
9fd325e25b | ||
|
|
93af43419c | ||
|
|
c8f0cdb74a | ||
|
|
19368fe5cc | ||
|
|
6a05e0eb6c | ||
|
|
40d643f11c | ||
|
|
72009cd21b | ||
|
|
cc9a50f40b | ||
|
|
e3135e8875 | ||
|
|
59eb297151 | ||
|
|
32b0c1c89e | ||
|
|
3328a8190d | ||
|
|
7aa6acdca6 | ||
|
|
7926ff1264 | ||
|
|
67b926dfb4 | ||
|
|
327f9e7a4b | ||
|
|
8be220d089 | ||
|
|
2dad40f49f | ||
|
|
99b724028f | ||
|
|
909fbcb0d4 | ||
|
|
351fc3e4d3 | ||
|
|
252b3d31a3 | ||
|
|
bfdfaab38c | ||
|
|
21a5963f84 | ||
|
|
9452249dce | ||
|
|
49a9df058f | ||
|
|
f99b7e5d17 | ||
|
|
ac03c57a94 | ||
|
|
2e4cf648c5 | ||
|
|
f000baa328 | ||
|
|
113f89604a | ||
|
|
5019a004ce | ||
|
|
a577f3844a | ||
|
|
88c7f0f690 | ||
|
|
a18792499a | ||
|
|
0b5fc8bb3c | ||
|
|
1f77410fda | ||
|
|
45fb57f29c | ||
|
|
3380b394eb | ||
|
|
1286cc5044 | ||
|
|
1ce1af791c | ||
|
|
2587e329ee | ||
|
|
37d4816051 | ||
|
|
d9dff882b8 | ||
|
|
26cc2c5278 | ||
|
|
69aff1bdc4 | ||
|
|
567994f2b1 | ||
|
|
e316c7b8aa | ||
|
|
c07a059a8b | ||
|
|
7b4bbffc41 | ||
|
|
bdc011ef34 | ||
|
|
44ea0d7c61 | ||
|
|
aa950e5ee4 | ||
|
|
247906e50e | ||
|
|
81b2493a44 | ||
|
|
97b0edb222 | ||
|
|
f0a9e4b9fe | ||
|
|
f35bbcaec5 | ||
|
|
2822061dc9 | ||
|
|
be481c8d17 | ||
|
|
605a972122 | ||
|
|
66d98d9fe7 | ||
|
|
6445ed37c9 | ||
|
|
ca745aeee8 | ||
|
|
f806b4927b | ||
|
|
2942eb5d7f | ||
|
|
2a5d6650ee | ||
|
|
43e971f57a | ||
|
|
785779613b | ||
|
|
481193f0ca | ||
|
|
2cb3cccd14 | ||
|
|
7c16766994 | ||
|
|
448d18deca | ||
|
|
97022b0733 | ||
|
|
c2a3e2d4cd | ||
|
|
b2dfcf17c8 | ||
|
|
d061f09281 | ||
|
|
daa64efd40 | ||
|
|
a37deabe27 | ||
|
|
6009ef0def | ||
|
|
edc644b0d0 | ||
|
|
1297af8baf | ||
|
|
b8b1b6675b | ||
|
|
7d3b7abc44 | ||
|
|
9a572f298e | ||
|
|
b53ca9e678 | ||
|
|
9aceec161a | ||
|
|
98b186ed55 | ||
|
|
f23ee1b5a8 | ||
|
|
c6d779f1fc | ||
|
|
8908b27b08 | ||
|
|
8947c9b116 | ||
|
|
9464caac6e | ||
|
|
98ce91c575 | ||
|
|
07f8feda3d | ||
|
|
a4e0496ff5 | ||
|
|
cf162c02c8 | ||
|
|
831aae94df | ||
|
|
3fd9f28778 | ||
|
|
a656e8e2a8 | ||
|
|
2f02152703 | ||
|
|
9d08f8ce67 | ||
|
|
ec6d508793 | ||
|
|
772e35ea75 | ||
|
|
1d28f886c8 | ||
|
|
d3dc8aeb9a | ||
|
|
a07d762934 | ||
|
|
85ac9e127d | ||
|
|
011c2823ff | ||
|
|
f28a94f03f | ||
|
|
5ecfc80cb9 | ||
|
|
e5e36ba050 | ||
|
|
0ad9116d6a | ||
|
|
0516032443 | ||
|
|
95d211c90f | ||
|
|
12d6a75ef7 | ||
|
|
06153dc373 | ||
|
|
02afa91fc3 | ||
|
|
c8b212789f | ||
|
|
95256d3fcf | ||
|
|
8a6d174c99 | ||
|
|
d36cf7f662 | ||
|
|
1b02a542c8 | ||
|
|
45430bb010 | ||
|
|
be2a139ade | ||
|
|
70d77b24f6 | ||
|
|
50c25d6d7b | ||
|
|
5bcdc0bc64 | ||
|
|
eff0f95b58 | ||
|
|
50b53d31bd | ||
|
|
b5a56852f3 | ||
|
|
b7486e34ad | ||
|
|
3edc5f1425 | ||
|
|
65b4b73cb5 | ||
|
|
186c08e8c3 | ||
|
|
7721aa0def | ||
|
|
4987c60e03 | ||
|
|
cd845f7fdd | ||
|
|
9e84d9e782 | ||
|
|
23c7fcc97f | ||
|
|
e4024efbc7 | ||
|
|
c4d53108af | ||
|
|
be8fe3564d | ||
|
|
1f95775057 | ||
|
|
f2a4dd875e | ||
|
|
67dd2300c8 | ||
|
|
b5391b06b4 | ||
|
|
54f2c71c13 | ||
|
|
6697900126 | ||
|
|
9ff3400b44 | ||
|
|
3ff0726ebf | ||
|
|
c96c939dfe | ||
|
|
719cf280c9 | ||
|
|
4ec6e2df04 | ||
|
|
031a9190c3 | ||
|
|
5204dcf327 | ||
|
|
cad623ef84 | ||
|
|
a59f58f8b6 | ||
|
|
d89c613f5d | ||
|
|
fa265ddb2f | ||
|
|
ec3dd04935 | ||
|
|
0c83e81410 | ||
|
|
7808a843cc | ||
|
|
615d7706af | ||
|
|
f05ca4d06a | ||
|
|
cb4145e2b6 | ||
|
|
6917c9aa7b | ||
|
|
c1d2451656 | ||
|
|
c41ec1fabd | ||
|
|
273c91882e | ||
|
|
ad6b5e5830 | ||
|
|
d441eb9d7a | ||
|
|
139d805c9f | ||
|
|
783347fc8e | ||
|
|
6d722d081d | ||
|
|
33a8c6ca0e | ||
|
|
ae043400f5 | ||
|
|
e41f96b31f | ||
|
|
7220f15158 | ||
|
|
b8d7eaa767 | ||
|
|
a612b463a8 | ||
|
|
493e7e81b5 | ||
|
|
b6cf0dbefd | ||
|
|
5d5c08f9e7 | ||
|
|
bfdfac0f81 | ||
|
|
9c72b7e3e7 | ||
|
|
46519e5c64 | ||
|
|
0ff961e203 | ||
|
|
cbc17c6832 | ||
|
|
93e5b15cba | ||
|
|
081e65d076 | ||
|
|
4ab2cfb713 | ||
|
|
a73335c0af | ||
|
|
0f3e257773 | ||
|
|
57d734d84f | ||
|
|
07f2e9c999 | ||
|
|
401883cae3 | ||
|
|
7bc92e1e3c | ||
|
|
de5f8b0653 | ||
|
|
cfa73ed53b | ||
|
|
67370bb1c7 | ||
|
|
f60593255a | ||
|
|
c6f9b97615 | ||
|
|
6161a394c2 | ||
|
|
b35cd42015 | ||
|
|
5e323993db | ||
|
|
f1623e419e | ||
|
|
a3047fb1bb | ||
|
|
b4d02f486e | ||
|
|
9e24893b9b | ||
|
|
47dec6ecef | ||
|
|
059fea672c | ||
|
|
389e804426 | ||
|
|
087a638c18 | ||
|
|
97757c74ca | ||
|
|
08f3ad583b | ||
|
|
16f01d31d3 | ||
|
|
e0f6c66351 | ||
|
|
3733b28772 | ||
|
|
24f8912134 | ||
|
|
cdc8847f9b | ||
|
|
25bd4b9eb5 | ||
|
|
d476191252 | ||
|
|
bf35d6a07c | ||
|
|
5378d38a05 | ||
|
|
85984f1173 | ||
|
|
92bb40349f | ||
|
|
fcee9bf738 | ||
|
|
cc2f3408ad | ||
|
|
89fb218041 | ||
|
|
4d0c3781e5 | ||
|
|
e2e319ac7f | ||
|
|
63508cad3f | ||
|
|
0b6eb6eed1 | ||
|
|
a41cfdeba5 | ||
|
|
7a6763ca33 | ||
|
|
2a6e19e4f9 | ||
|
|
9f9526b722 | ||
|
|
267684b7c3 | ||
|
|
b74dd3d576 | ||
|
|
278ae48842 | ||
|
|
189ca12627 | ||
|
|
5108f57ba8 | ||
|
|
6bace666e4 | ||
|
|
4c35616d30 | ||
|
|
cadbc733b8 | ||
|
|
ee6088957d | ||
|
|
58118c89f7 | ||
|
|
108112c9a0 | ||
|
|
4f586719e7 | ||
|
|
31e4985480 | ||
|
|
672b8bd262 | ||
|
|
c64c185407 | ||
|
|
19fd6aa4e3 | ||
|
|
55746d05f3 | ||
|
|
ad8d6382d4 | ||
|
|
5114a6c09c | ||
|
|
f405c109e2 | ||
|
|
ff646fee07 | ||
|
|
e517a8998c | ||
|
|
776a699a76 | ||
|
|
a942a63959 | ||
|
|
4b275e3370 | ||
|
|
7bb6787800 | ||
|
|
d314dda749 |
@@ -1,194 +0,0 @@
|
||||
# Copyright 2020 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# We want to use LTS ubuntu from our mirror because dockerhub has a
|
||||
# rate limit.
|
||||
# FROM mirror.gcr.io/library/ubuntu:18.04
|
||||
# However, now the above image is not working, we're using our own cache
|
||||
FROM gcr.io/cloud-devrel-kokoro-resources/ubuntu:20.04
|
||||
|
||||
ENV DEBIAN_FRONTEND noninteractive
|
||||
|
||||
# Ensure local Python is preferred over distribution Python.
|
||||
ENV PATH /usr/local/bin:$PATH
|
||||
|
||||
# http://bugs.python.org/issue19846
|
||||
# At the moment, setting "LANG=C" on a Linux system fundamentally breaks
|
||||
# Python 3.
|
||||
ENV LANG C.UTF-8
|
||||
|
||||
# Install dependencies.
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
apt-transport-https \
|
||||
build-essential \
|
||||
ca-certificates \
|
||||
curl \
|
||||
dirmngr \
|
||||
git \
|
||||
gcc \
|
||||
gpg-agent \
|
||||
graphviz \
|
||||
libbz2-dev \
|
||||
libdb5.3-dev \
|
||||
libexpat1-dev \
|
||||
libffi-dev \
|
||||
liblzma-dev \
|
||||
libmagickwand-dev \
|
||||
libmemcached-dev \
|
||||
libpython3-dev \
|
||||
libreadline-dev \
|
||||
libsnappy-dev \
|
||||
libssl-dev \
|
||||
libsqlite3-dev \
|
||||
portaudio19-dev \
|
||||
pkg-config \
|
||||
redis-server \
|
||||
software-properties-common \
|
||||
ssh \
|
||||
sudo \
|
||||
systemd \
|
||||
tcl \
|
||||
tcl-dev \
|
||||
tk \
|
||||
tk-dev \
|
||||
uuid-dev \
|
||||
wget \
|
||||
zlib1g-dev \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
# Install docker
|
||||
RUN curl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo apt-key add -
|
||||
|
||||
RUN add-apt-repository \
|
||||
"deb [arch=amd64] https://download.docker.com/linux/ubuntu \
|
||||
$(lsb_release -cs) \
|
||||
stable"
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
docker-ce \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
# Install Bazel for compiling Tink in Cloud SQL Client Side Encryption Samples
|
||||
# TODO: Delete this section once google/tink#483 is resolved
|
||||
RUN apt install -y curl gpgconf gpg \
|
||||
&& curl -fsSL https://bazel.build/bazel-release.pub.gpg | gpg --dearmor > bazel.gpg \
|
||||
&& mv bazel.gpg /etc/apt/trusted.gpg.d/ \
|
||||
&& echo "deb [arch=amd64] https://storage.googleapis.com/bazel-apt stable jdk1.8" | sudo tee /etc/apt/sources.list.d/bazel.list \
|
||||
&& apt update && apt install -y bazel \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
# Install Microsoft ODBC 17 Driver and unixodbc for testing SQL Server samples
|
||||
RUN curl https://packages.microsoft.com/keys/microsoft.asc | apt-key add - \
|
||||
&& curl https://packages.microsoft.com/config/ubuntu/20.04/prod.list > /etc/apt/sources.list.d/mssql-release.list \
|
||||
&& apt-get update \
|
||||
&& ACCEPT_EULA=Y apt-get install -y --no-install-recommends \
|
||||
msodbcsql17 \
|
||||
unixodbc-dev \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
COPY fetch_gpg_keys.sh /tmp
|
||||
# Install the desired versions of Python.
|
||||
RUN set -ex \
|
||||
&& export GNUPGHOME="$(mktemp -d)" \
|
||||
&& echo "disable-ipv6" >> "${GNUPGHOME}/dirmngr.conf" \
|
||||
&& /tmp/fetch_gpg_keys.sh \
|
||||
&& for PYTHON_VERSION in 2.7.18 3.6.13 3.7.10 3.8.8 3.9.2; do \
|
||||
wget --no-check-certificate -O python-${PYTHON_VERSION}.tar.xz "https://www.python.org/ftp/python/${PYTHON_VERSION%%[a-z]*}/Python-$PYTHON_VERSION.tar.xz" \
|
||||
&& wget --no-check-certificate -O python-${PYTHON_VERSION}.tar.xz.asc "https://www.python.org/ftp/python/${PYTHON_VERSION%%[a-z]*}/Python-$PYTHON_VERSION.tar.xz.asc" \
|
||||
&& gpg --batch --verify python-${PYTHON_VERSION}.tar.xz.asc python-${PYTHON_VERSION}.tar.xz \
|
||||
&& rm -r python-${PYTHON_VERSION}.tar.xz.asc \
|
||||
&& mkdir -p /usr/src/python-${PYTHON_VERSION} \
|
||||
&& tar -xJC /usr/src/python-${PYTHON_VERSION} --strip-components=1 -f python-${PYTHON_VERSION}.tar.xz \
|
||||
&& rm python-${PYTHON_VERSION}.tar.xz \
|
||||
&& cd /usr/src/python-${PYTHON_VERSION} \
|
||||
&& ./configure \
|
||||
--enable-shared \
|
||||
# This works only on Python 2.7 and throws a warning on every other
|
||||
# version, but seems otherwise harmless.
|
||||
--enable-unicode=ucs4 \
|
||||
--with-system-ffi \
|
||||
--without-ensurepip \
|
||||
&& make -j$(nproc) \
|
||||
&& make install \
|
||||
&& ldconfig \
|
||||
; done \
|
||||
&& rm -rf "${GNUPGHOME}" \
|
||||
&& rm -rf /usr/src/python* \
|
||||
&& rm -rf ~/.cache/
|
||||
|
||||
|
||||
# Install pip on Python 3.6 only.
|
||||
# If the environment variable is called "PIP_VERSION", pip explodes with
|
||||
# "ValueError: invalid truth value '<VERSION>'"
|
||||
ENV PYTHON_PIP_VERSION 20.2.4
|
||||
RUN wget --no-check-certificate -O /tmp/get-pip.py 'https://bootstrap.pypa.io/get-pip.py' \
|
||||
&& python3.6 /tmp/get-pip.py "pip==$PYTHON_PIP_VERSION" \
|
||||
# we use "--force-reinstall" for the case where the version of pip we're trying to install is the same as the version bundled with Python
|
||||
# ("Requirement already up-to-date: pip==8.1.2 in /usr/local/lib/python3.6/site-packages")
|
||||
# https://github.com/docker-library/python/pull/143#issuecomment-241032683
|
||||
&& pip3 install --no-cache-dir --upgrade --force-reinstall "pip==$PYTHON_PIP_VERSION" \
|
||||
# then we use "pip list" to ensure we don't have more than one pip version installed
|
||||
# https://github.com/docker-library/python/pull/100
|
||||
&& [ "$(pip list |tac|tac| awk -F '[ ()]+' '$1 == "pip" { print $2; exit }')" = "$PYTHON_PIP_VERSION" ]
|
||||
|
||||
# Ensure Pip for python3
|
||||
RUN python3 /tmp/get-pip.py
|
||||
RUN rm /tmp/get-pip.py
|
||||
|
||||
# Install "virtualenv", since the vast majority of users of this image
|
||||
# will want it.
|
||||
RUN pip install --no-cache-dir virtualenv
|
||||
|
||||
# Setup Cloud SDK
|
||||
ENV CLOUD_SDK_VERSION 339.0.0
|
||||
# Use system python for cloud sdk.
|
||||
ENV CLOUDSDK_PYTHON python3.6
|
||||
RUN wget https://dl.google.com/dl/cloudsdk/channels/rapid/downloads/google-cloud-sdk-$CLOUD_SDK_VERSION-linux-x86_64.tar.gz
|
||||
RUN tar xzf google-cloud-sdk-$CLOUD_SDK_VERSION-linux-x86_64.tar.gz
|
||||
RUN /google-cloud-sdk/install.sh
|
||||
ENV PATH /google-cloud-sdk/bin:$PATH
|
||||
|
||||
# Enable redis-server on boot.
|
||||
RUN sudo systemctl enable redis-server.service
|
||||
|
||||
# Create a user and allow sudo
|
||||
|
||||
# kbuilder uid on the default Kokoro image
|
||||
ARG UID=1000
|
||||
ARG USERNAME=kbuilder
|
||||
|
||||
# Add a new user to the container image.
|
||||
# This is needed for ssh and sudo access.
|
||||
|
||||
# Add a new user with the caller's uid and the username.
|
||||
RUN useradd -d /h -u ${UID} ${USERNAME}
|
||||
|
||||
# Allow nopasswd sudo
|
||||
RUN echo "${USERNAME} ALL=(ALL) NOPASSWD:ALL" >> /etc/sudoers
|
||||
|
||||
CMD ["python3.6"]
|
||||
@@ -1,11 +1,14 @@
|
||||
from typing import List
|
||||
from ratemate import RateLimit
|
||||
from resource_cleanup_manager import (
|
||||
ResourceCleanupManager,
|
||||
DatasetResourceCleanupManager,
|
||||
EndpointResourceCleanupManager,
|
||||
ModelResourceCleanupManager,
|
||||
EndpointResourceCleanupManager,
|
||||
ResourceCleanupManager,
|
||||
)
|
||||
|
||||
rate_limit = RateLimit(max_count=25, per=60, greedy=False)
|
||||
|
||||
|
||||
def run_cleanup_managers(managers: List[ResourceCleanupManager], is_dry_run: bool):
|
||||
for manager in managers:
|
||||
@@ -15,14 +18,18 @@ def run_cleanup_managers(managers: List[ResourceCleanupManager], is_dry_run: boo
|
||||
resources = manager.list()
|
||||
print(f"Found {len(resources)} {type_name}'s")
|
||||
for resource in resources:
|
||||
if not manager.is_deletable(resource):
|
||||
continue
|
||||
try:
|
||||
if not manager.is_deletable(resource):
|
||||
continue
|
||||
|
||||
if is_dry_run:
|
||||
resource_name = manager.resource_name(resource)
|
||||
print(f"Will delete '{type_name}': {resource_name}")
|
||||
else:
|
||||
manager.delete(resource)
|
||||
if is_dry_run:
|
||||
resource_name = manager.resource_name(resource)
|
||||
print(f"Will delete '{type_name}': {resource_name}")
|
||||
else:
|
||||
rate_limit.wait() # wait before deleting
|
||||
manager.delete(resource)
|
||||
except Exception as exception:
|
||||
print(exception)
|
||||
|
||||
print("")
|
||||
|
||||
@@ -36,7 +43,7 @@ if is_dry_run:
|
||||
managers = [
|
||||
DatasetResourceCleanupManager(),
|
||||
EndpointResourceCleanupManager(),
|
||||
ModelResourceCleanupManager(),
|
||||
ModelResourceCleanupManager(), # ModelResourceCleanupManager must follow EndpointResourceCleanupManager due to deployed models blocking model deletion.
|
||||
]
|
||||
|
||||
run_cleanup_managers(managers=managers, is_dry_run=is_dry_run)
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
import abc
|
||||
from typing import Any, Type
|
||||
|
||||
from google.cloud import aiplatform
|
||||
from typing import Any
|
||||
from proto.datetime_helpers import DatetimeWithNanoseconds
|
||||
from google.cloud.aiplatform import base
|
||||
from proto.datetime_helpers import DatetimeWithNanoseconds
|
||||
|
||||
# If a resource was updated within this number of seconds, do not delete.
|
||||
RESOURCE_UPDATE_BUFFER_IN_SECONDS = 60 * 60 * 8
|
||||
@@ -40,7 +41,7 @@ class ResourceCleanupManager(abc.ABC):
|
||||
# Check that it wasn't created too recently, to prevent race conditions
|
||||
if time_difference <= RESOURCE_UPDATE_BUFFER_IN_SECONDS:
|
||||
print(
|
||||
f"Skipping '{resource}' due update_time being '{time_difference}', which is less than '{RESOURCE_UPDATE_BUFFER_IN_SECONDS}'."
|
||||
f"Skipping '{resource}' due to update_time being '{time_difference}', which is less than '{RESOURCE_UPDATE_BUFFER_IN_SECONDS}'."
|
||||
)
|
||||
return False
|
||||
|
||||
@@ -50,7 +51,7 @@ class ResourceCleanupManager(abc.ABC):
|
||||
class VertexAIResourceCleanupManager(ResourceCleanupManager):
|
||||
@property
|
||||
@abc.abstractmethod
|
||||
def vertex_ai_resource(self) -> base.VertexAiResourceNounWithFutureManager:
|
||||
def vertex_ai_resource(self) -> Type[base.VertexAiResourceNounWithFutureManager]:
|
||||
pass
|
||||
|
||||
@property
|
||||
@@ -60,7 +61,9 @@ class VertexAIResourceCleanupManager(ResourceCleanupManager):
|
||||
def list(self) -> Any:
|
||||
return self.vertex_ai_resource.list()
|
||||
|
||||
def resource_name(self, resource: Any) -> str:
|
||||
def resource_name(
|
||||
self, resource: Type[base.VertexAiResourceNounWithFutureManager]
|
||||
) -> str:
|
||||
return resource.display_name
|
||||
|
||||
def delete(self, resource):
|
||||
@@ -74,12 +77,33 @@ class VertexAIResourceCleanupManager(ResourceCleanupManager):
|
||||
|
||||
class DatasetResourceCleanupManager(VertexAIResourceCleanupManager):
|
||||
vertex_ai_resource = aiplatform.datasets._Dataset
|
||||
dataset_types = [
|
||||
aiplatform.ImageDataset,
|
||||
aiplatform.TabularDataset,
|
||||
aiplatform.TextDataset,
|
||||
aiplatform.TimeSeriesDataset,
|
||||
aiplatform.VideoDataset,
|
||||
]
|
||||
|
||||
def list(self) -> Any:
|
||||
return [
|
||||
dataset
|
||||
for dataset_type in self.dataset_types
|
||||
for dataset in dataset_type.list()
|
||||
]
|
||||
|
||||
|
||||
class EndpointResourceCleanupManager(VertexAIResourceCleanupManager):
|
||||
vertex_ai_resource = aiplatform.Endpoint
|
||||
|
||||
def delete(self, resource):
|
||||
# TODO: Remove this once https://github.com/googleapis/python-aiplatform/issues/1441 is fixed
|
||||
resource._sync_gca_resource()
|
||||
for deployed_model_id in [
|
||||
models.id for models in resource._gca_resource.deployed_models
|
||||
]:
|
||||
resource._undeploy(deployed_model_id=deployed_model_id)
|
||||
|
||||
resource.delete(force=True)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env python
|
||||
# Copyright 2021 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""A CLI to process changed notebooks and execute them on Google Cloud Build"""
|
||||
|
||||
import argparse
|
||||
import pathlib
|
||||
|
||||
import execute_changed_notebooks_helper
|
||||
|
||||
|
||||
def str2bool(v):
|
||||
if isinstance(v, bool):
|
||||
return v
|
||||
if v.lower() in ("yes", "true", "t", "y", "1"):
|
||||
return True
|
||||
elif v.lower() in ("no", "false", "f", "n", "0"):
|
||||
return False
|
||||
else:
|
||||
raise argparse.ArgumentTypeError("Boolean value expected.")
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
parser.add_argument(
|
||||
"--test_paths_file",
|
||||
type=pathlib.Path,
|
||||
help="The path to the file that has newline-limited folders of notebooks that should be tested.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--base_branch",
|
||||
help="The base git branch to diff against to find changed files.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container_uri",
|
||||
type=str,
|
||||
help="The container uri to run each notebook in.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_project_id",
|
||||
type=str,
|
||||
help="The GCP project id. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_region",
|
||||
type=str,
|
||||
help="The GCP region. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--staging_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for staging temporary files.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifacts_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for storing executed notebooks.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--timeout",
|
||||
type=int,
|
||||
help="Timeout in seconds",
|
||||
default=86400,
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--private_pool_id",
|
||||
type=str,
|
||||
help="The private pool id.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--should_parallelize",
|
||||
type=str2bool,
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=True,
|
||||
help="Should run notebooks in parallel.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
notebooks = execute_changed_notebooks_helper.get_changed_notebooks(
|
||||
test_paths_file=args.test_paths_file,
|
||||
base_branch=args.base_branch,
|
||||
)
|
||||
|
||||
execute_changed_notebooks_helper.process_and_execute_notebooks(
|
||||
notebooks=notebooks,
|
||||
container_uri=args.container_uri,
|
||||
staging_bucket=args.staging_bucket,
|
||||
artifacts_bucket=args.artifacts_bucket,
|
||||
variable_project_id=args.variable_project_id,
|
||||
variable_region=args.variable_region,
|
||||
private_pool_id=args.private_pool_id,
|
||||
should_parallelize=args.should_parallelize,
|
||||
timeout=args.timeout,
|
||||
)
|
||||
@@ -13,34 +13,28 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import concurrent
|
||||
import dataclasses
|
||||
import datetime
|
||||
import functools
|
||||
import git
|
||||
import operator
|
||||
import os
|
||||
import pathlib
|
||||
import nbformat
|
||||
import re
|
||||
import subprocess
|
||||
from typing import List, Optional
|
||||
from tabulate import tabulate
|
||||
import operator
|
||||
|
||||
import execute_notebook_helper
|
||||
import execute_notebook_remote
|
||||
from utils import util, NotebookProcessors
|
||||
import nbformat
|
||||
from google.cloud.devtools.cloudbuild_v1.types import BuildOperationMetadata
|
||||
from ratemate import RateLimit
|
||||
from tabulate import tabulate
|
||||
from utils import NotebookProcessors, util
|
||||
|
||||
|
||||
def str2bool(v):
|
||||
if isinstance(v, bool):
|
||||
return v
|
||||
if v.lower() in ("yes", "true", "t", "y", "1"):
|
||||
return True
|
||||
elif v.lower() in ("no", "false", "f", "n", "0"):
|
||||
return False
|
||||
else:
|
||||
raise argparse.ArgumentTypeError("Boolean value expected.")
|
||||
# A buffer so that workers finish before the orchestrating job
|
||||
WORKER_TIMEOUT_BUFFER_IN_SECONDS: int = 60 * 60
|
||||
|
||||
|
||||
def format_timedelta(delta: datetime.timedelta) -> str:
|
||||
@@ -115,15 +109,22 @@ def _create_tag(filepath: str) -> str:
|
||||
return tag
|
||||
|
||||
|
||||
def execute_notebook(
|
||||
rate_limit = RateLimit(max_count=50, per=60, greedy=True)
|
||||
|
||||
|
||||
def process_and_execute_notebook(
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
private_pool_id: Optional[str],
|
||||
deadline: datetime,
|
||||
notebook: str,
|
||||
should_get_tail_logs: bool = False,
|
||||
) -> NotebookExecutionResult:
|
||||
rate_limit.wait() # wait before creating the task
|
||||
|
||||
print(f"Running notebook: {notebook}")
|
||||
|
||||
# Create paths
|
||||
@@ -156,12 +157,20 @@ def execute_notebook(
|
||||
# Upload the pre-processed code to a GCS bucket
|
||||
code_archive_uri = util.archive_code_and_upload(staging_bucket=staging_bucket)
|
||||
|
||||
# Calculate timeout in seconds
|
||||
timeout_in_seconds = max(
|
||||
int((deadline - datetime.datetime.now()).total_seconds()), 1
|
||||
)
|
||||
|
||||
operation = execute_notebook_remote.execute_notebook_remote(
|
||||
code_archive_uri=code_archive_uri,
|
||||
notebook_uri=notebook,
|
||||
notebook_output_uri=notebook_output_uri,
|
||||
container_uri=container_uri,
|
||||
tag=tag,
|
||||
private_pool_id=private_pool_id,
|
||||
private_pool_region=variable_region,
|
||||
timeout_in_seconds=timeout_in_seconds,
|
||||
)
|
||||
|
||||
operation_metadata = BuildOperationMetadata(mapping=operation.metadata)
|
||||
@@ -202,15 +211,74 @@ def execute_notebook(
|
||||
return result
|
||||
|
||||
|
||||
def run_changed_notebooks(
|
||||
def get_changed_notebooks(
|
||||
test_paths_file: str,
|
||||
base_branch: Optional[str] = None,
|
||||
) -> List[str]:
|
||||
"""
|
||||
Get the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only returns notebooks that have differences from the Git base_branch.
|
||||
"""
|
||||
|
||||
test_paths = []
|
||||
with open(test_paths_file) as file:
|
||||
lines = [line.strip() for line in file.readlines()]
|
||||
lines = [line for line in lines if len(line) > 0]
|
||||
test_paths = [line for line in lines]
|
||||
|
||||
if len(test_paths) == 0:
|
||||
raise RuntimeError("No test folders found.")
|
||||
|
||||
print(f"Checking folders: {test_paths}")
|
||||
|
||||
# Find notebooks
|
||||
notebooks = []
|
||||
|
||||
# Instantiate GitPython objects
|
||||
repo = git.Repo(os.getcwd())
|
||||
index = repo.index
|
||||
|
||||
if base_branch:
|
||||
# Get the point at which this branch branches off from main
|
||||
branching_commits = repo.merge_base("HEAD", f"origin/{base_branch}")
|
||||
|
||||
if len(branching_commits) > 0:
|
||||
branching_commit = branching_commits[0]
|
||||
print(f"Looking for notebooks that changed from branch: {branching_commit}")
|
||||
|
||||
notebooks = [
|
||||
diff.b_path
|
||||
for diff in index.diff(branching_commit, paths=test_paths)
|
||||
if diff.b_path is not None
|
||||
]
|
||||
else:
|
||||
notebooks = []
|
||||
else:
|
||||
print(f"Looking for all notebooks.")
|
||||
notebooks = subprocess.check_output(["git", "ls-files"] + test_paths)
|
||||
|
||||
notebooks = [notebook for notebook in notebooks if notebook.endswith(".ipynb")]
|
||||
notebooks = [notebook for notebook in notebooks if len(notebook) > 0]
|
||||
notebooks = [notebook for notebook in notebooks if pathlib.Path(notebook).exists()]
|
||||
|
||||
if len(notebooks) > 0:
|
||||
print(f"Found {len(notebooks)} notebooks:")
|
||||
for notebook in notebooks:
|
||||
print(f"\t{notebook}")
|
||||
|
||||
return notebooks
|
||||
|
||||
|
||||
def process_and_execute_notebooks(
|
||||
notebooks: List[str],
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
private_pool_id: Optional[str],
|
||||
should_parallelize: bool,
|
||||
base_branch: Optional[str] = None,
|
||||
timeout: int,
|
||||
):
|
||||
"""
|
||||
Run the notebooks that exist under the folders defined in the test_paths_file.
|
||||
@@ -237,171 +305,114 @@ def run_changed_notebooks(
|
||||
Required. The value for REGION to inject into notebooks.
|
||||
should_parallelize (bool):
|
||||
Required. Should run notebooks in parallel using a thread pool as opposed to in sequence.
|
||||
timeout (str):
|
||||
Required. Timeout string according to https://cloud.google.com/build/docs/build-config-file-schema#timeout.
|
||||
"""
|
||||
|
||||
test_paths = []
|
||||
with open(test_paths_file) as file:
|
||||
lines = [line.strip() for line in file.readlines()]
|
||||
lines = [line for line in lines if len(line) > 0]
|
||||
test_paths = [line for line in lines]
|
||||
# Calculate deadline
|
||||
deadline = datetime.datetime.now() + datetime.timedelta(
|
||||
seconds=max(timeout - WORKER_TIMEOUT_BUFFER_IN_SECONDS, 0)
|
||||
)
|
||||
|
||||
if len(test_paths) == 0:
|
||||
raise RuntimeError("No test folders found.")
|
||||
if len(notebooks) > 1:
|
||||
notebook_execution_results: List[NotebookExecutionResult] = []
|
||||
|
||||
print(f"Checking folders: {test_paths}")
|
||||
|
||||
# Find notebooks
|
||||
notebooks = []
|
||||
if base_branch:
|
||||
print(f"Looking for notebooks that changed from branch: {base_branch}")
|
||||
notebooks = subprocess.check_output(
|
||||
["git", "diff", "--name-only", f"origin/{base_branch}..."] + test_paths
|
||||
)
|
||||
else:
|
||||
print(f"Looking for all notebooks.")
|
||||
notebooks = subprocess.check_output(["git", "ls-files"] + test_paths)
|
||||
|
||||
notebooks = notebooks.decode("utf-8").split("\n")
|
||||
notebooks = [notebook for notebook in notebooks if notebook.endswith(".ipynb")]
|
||||
notebooks = [notebook for notebook in notebooks if len(notebook) > 0]
|
||||
notebooks = [notebook for notebook in notebooks if pathlib.Path(notebook).exists()]
|
||||
|
||||
notebook_execution_results: List[NotebookExecutionResult] = []
|
||||
|
||||
if len(notebooks) > 0:
|
||||
print(f"Found {len(notebooks)} modified notebooks: {notebooks}")
|
||||
|
||||
if should_parallelize and len(notebooks) > 1:
|
||||
print(
|
||||
"Running notebooks in parallel, so no logs will be displayed. Please wait..."
|
||||
)
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=None) as executor:
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=100) as executor:
|
||||
print(f"Max workers: {executor._max_workers}")
|
||||
|
||||
notebook_execution_results = list(
|
||||
executor.map(
|
||||
functools.partial(
|
||||
execute_notebook,
|
||||
process_and_execute_notebook,
|
||||
container_uri,
|
||||
staging_bucket,
|
||||
artifacts_bucket,
|
||||
variable_project_id,
|
||||
variable_region,
|
||||
private_pool_id,
|
||||
deadline,
|
||||
),
|
||||
notebooks,
|
||||
)
|
||||
)
|
||||
else:
|
||||
notebook_execution_results = [
|
||||
execute_notebook(
|
||||
process_and_execute_notebook(
|
||||
container_uri=container_uri,
|
||||
staging_bucket=staging_bucket,
|
||||
artifacts_bucket=artifacts_bucket,
|
||||
variable_project_id=variable_project_id,
|
||||
variable_region=variable_region,
|
||||
private_pool_id=private_pool_id,
|
||||
deadline=deadline,
|
||||
notebook=notebook,
|
||||
)
|
||||
for notebook in notebooks
|
||||
]
|
||||
|
||||
print("\n=== RESULTS ===\n")
|
||||
|
||||
results_sorted = sorted(
|
||||
notebook_execution_results,
|
||||
key=lambda result: result.is_pass,
|
||||
reverse=True,
|
||||
)
|
||||
|
||||
# Print results
|
||||
print(
|
||||
tabulate(
|
||||
[
|
||||
[
|
||||
result.name,
|
||||
"PASSED" if result.is_pass else "FAILED",
|
||||
format_timedelta(result.duration),
|
||||
result.log_url,
|
||||
result.output_uri,
|
||||
]
|
||||
for result in results_sorted
|
||||
],
|
||||
headers=["build_tag", "status", "duration", "log_url", "output_url"],
|
||||
)
|
||||
)
|
||||
|
||||
print("\n=== END RESULTS===\n")
|
||||
|
||||
total_notebook_duration = functools.reduce(
|
||||
operator.add,
|
||||
[datetime.timedelta(seconds=0)]
|
||||
+ [result.duration for result in results_sorted],
|
||||
)
|
||||
|
||||
print(
|
||||
f"Cumulative notebook duration: {format_timedelta(total_notebook_duration)}"
|
||||
)
|
||||
|
||||
# Raise error if any notebooks failed
|
||||
if not all([result.is_pass for result in results_sorted]):
|
||||
raise RuntimeError("Notebook failures detected. See logs for details")
|
||||
|
||||
elif len(notebooks) == 1:
|
||||
notebook = notebooks[0]
|
||||
|
||||
# Pre-process notebook by substituting variable names
|
||||
_process_notebook(
|
||||
notebook_path=notebook,
|
||||
variable_project_id=variable_project_id,
|
||||
variable_region=variable_region,
|
||||
)
|
||||
|
||||
execute_notebook_helper.execute_notebook(
|
||||
notebook_source=notebook,
|
||||
output_file_or_uri="/".join(
|
||||
[artifacts_bucket, pathlib.Path(notebook).name]
|
||||
),
|
||||
should_log_output=True,
|
||||
)
|
||||
else:
|
||||
print("No notebooks modified in this pull request.")
|
||||
|
||||
print("\n=== RESULTS ===\n")
|
||||
|
||||
results_sorted = sorted(
|
||||
notebook_execution_results,
|
||||
key=lambda result: result.is_pass,
|
||||
reverse=True,
|
||||
)
|
||||
|
||||
# Print results
|
||||
print(
|
||||
tabulate(
|
||||
[
|
||||
[
|
||||
result.name,
|
||||
"PASSED" if result.is_pass else "FAILED",
|
||||
format_timedelta(result.duration),
|
||||
result.log_url,
|
||||
]
|
||||
for result in results_sorted
|
||||
],
|
||||
headers=["build_tag", "status", "duration", "log_url"],
|
||||
)
|
||||
)
|
||||
|
||||
print("\n=== END RESULTS===\n")
|
||||
|
||||
total_notebook_duration = functools.reduce(
|
||||
operator.add,
|
||||
[datetime.timedelta(seconds=0)]
|
||||
+ [result.duration for result in results_sorted],
|
||||
)
|
||||
|
||||
print(f"Cumulative notebook duration: {format_timedelta(total_notebook_duration)}")
|
||||
|
||||
# Raise error if any notebooks failed
|
||||
if not all([result.is_pass for result in results_sorted]):
|
||||
raise RuntimeError("Notebook failures detected. See logs for details")
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
parser.add_argument(
|
||||
"--test_paths_file",
|
||||
type=pathlib.Path,
|
||||
help="The path to the file that has newline-limited folders of notebooks that should be tested.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--base_branch",
|
||||
help="The base git branch to diff against to find changed files.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container_uri",
|
||||
type=str,
|
||||
help="The container uri to run each notebook in.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_project_id",
|
||||
type=str,
|
||||
help="The GCP project id. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_region",
|
||||
type=str,
|
||||
help="The GCP region. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--staging_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for staging temporary files.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifacts_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for storing executed notebooks.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--should_parallelize",
|
||||
type=str2bool,
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=True,
|
||||
help="Should run notebooks in parallel.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
run_changed_notebooks(
|
||||
test_paths_file=args.test_paths_file,
|
||||
container_uri=args.container_uri,
|
||||
staging_bucket=args.staging_bucket,
|
||||
artifacts_bucket=args.artifacts_bucket,
|
||||
variable_project_id=args.variable_project_id,
|
||||
variable_region=args.variable_region,
|
||||
should_parallelize=args.should_parallelize,
|
||||
base_branch=args.base_branch,
|
||||
)
|
||||
@@ -13,10 +13,13 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import ExecuteNotebook
|
||||
"""A CLI to download (optional) and run a single notebook locally"""
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
import argparse
|
||||
|
||||
import execute_notebook_helper
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run a single notebook locally.")
|
||||
parser.add_argument(
|
||||
"--notebook_source",
|
||||
type=str,
|
||||
@@ -31,7 +34,7 @@ parser.add_argument(
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
ExecuteNotebook.execute_notebook(
|
||||
execute_notebook_helper.execute_notebook(
|
||||
notebook_source=args.notebook_source,
|
||||
output_file_or_uri=args.output_file_or_uri,
|
||||
should_log_output=True,
|
||||
|
||||
@@ -13,14 +13,16 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import sys
|
||||
import os
|
||||
import errno
|
||||
import papermill as pm
|
||||
import shutil
|
||||
"""Methods to run a notebook locally"""
|
||||
|
||||
from utils import util
|
||||
import errno
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
|
||||
import papermill as pm
|
||||
from google.cloud.aiplatform import utils
|
||||
from utils import util
|
||||
|
||||
# This script is used to execute a notebook and write out the output notebook.
|
||||
|
||||
@@ -30,6 +32,7 @@ def execute_notebook(
|
||||
output_file_or_uri: str,
|
||||
should_log_output: bool,
|
||||
):
|
||||
"""Execute a single notebook using Papermill"""
|
||||
file_name = os.path.basename(os.path.normpath(notebook_source))
|
||||
|
||||
# Download notebook if it's a GCS URI
|
||||
@@ -1,18 +1,34 @@
|
||||
#!/usr/bin/env python
|
||||
# Copyright 2021 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Methods to run a notebook on Google Cloud Build"""
|
||||
|
||||
from re import sub
|
||||
from typing import Optional
|
||||
|
||||
import google.auth
|
||||
import yaml
|
||||
from google.api_core import client_options, operation
|
||||
from google.cloud.aiplatform import utils
|
||||
from google.cloud.devtools import cloudbuild_v1
|
||||
from google.cloud.devtools.cloudbuild_v1.types import Source, StorageSource
|
||||
from google.protobuf import duration_pb2
|
||||
from yaml.loader import FullLoader
|
||||
|
||||
import google.auth
|
||||
from google.cloud.devtools import cloudbuild_v1
|
||||
from google.cloud.devtools.cloudbuild_v1.types import Source, StorageSource
|
||||
|
||||
from typing import Optional
|
||||
import yaml
|
||||
|
||||
from google.cloud.aiplatform import utils
|
||||
from google.api_core import operation
|
||||
|
||||
CLOUD_BUILD_FILEPATH = ".cloud-build/notebook-execution-test-cloudbuild-single.yaml"
|
||||
TIMEOUT_IN_SECONDS = 86400
|
||||
SERVICE_BASE_PATH = "cloudbuild.googleapis.com"
|
||||
|
||||
|
||||
def execute_notebook_remote(
|
||||
@@ -20,20 +36,14 @@ def execute_notebook_remote(
|
||||
notebook_uri: str,
|
||||
notebook_output_uri: str,
|
||||
container_uri: str,
|
||||
private_pool_id: Optional[str],
|
||||
private_pool_region: Optional[str],
|
||||
tag: Optional[str],
|
||||
timeout_in_seconds: Optional[int] = None,
|
||||
) -> operation.Operation:
|
||||
"""Create and execute a simple Google Cloud Build configuration,
|
||||
print the in-progress status and print the completed status."""
|
||||
"""Create and execute a single notebook on Google Cloud Build"""
|
||||
# Load build steps from YAML
|
||||
|
||||
# Authorize the client with Google defaults
|
||||
credentials, project_id = google.auth.default()
|
||||
client = cloudbuild_v1.services.cloud_build.CloudBuildClient()
|
||||
|
||||
build = cloudbuild_v1.Build()
|
||||
|
||||
# The following build steps will output "hello world"
|
||||
# For more information on build configuration, see
|
||||
# https://cloud.google.com/build/docs/configuring-builds/create-basic-configuration
|
||||
cloudbuild_config = yaml.load(open(CLOUD_BUILD_FILEPATH), Loader=FullLoader)
|
||||
|
||||
substitutions = {
|
||||
@@ -42,6 +52,24 @@ def execute_notebook_remote(
|
||||
"_NOTEBOOK_OUTPUT_GCS_URI": notebook_output_uri,
|
||||
}
|
||||
|
||||
build = cloudbuild_v1.Build()
|
||||
|
||||
options: Optional[client_options.ClientOptions] = None
|
||||
if private_pool_id and private_pool_region:
|
||||
# substitutions["_PRIVATE_POOL_NAME"] = private_pool_id
|
||||
build.options = cloudbuild_config.get("options")
|
||||
build.options.pool = {"name": private_pool_id}
|
||||
|
||||
# Switch to the regional endpoint of the pool
|
||||
options = client_options.ClientOptions(
|
||||
api_endpoint=f"{private_pool_region}-{SERVICE_BASE_PATH}"
|
||||
)
|
||||
|
||||
# Authorize the client with Google defaults
|
||||
credentials, project_id = google.auth.default()
|
||||
|
||||
client = cloudbuild_v1.services.cloud_build.CloudBuildClient(client_options=options)
|
||||
|
||||
(
|
||||
source_archived_file_gcs_bucket,
|
||||
source_archived_file_gcs_object,
|
||||
@@ -56,8 +84,8 @@ def execute_notebook_remote(
|
||||
|
||||
build.steps = cloudbuild_config["steps"]
|
||||
build.substitutions = substitutions
|
||||
build.timeout = duration_pb2.Duration(seconds=TIMEOUT_IN_SECONDS)
|
||||
build.queue_ttl = duration_pb2.Duration(seconds=TIMEOUT_IN_SECONDS)
|
||||
build.timeout = duration_pb2.Duration(seconds=timeout_in_seconds)
|
||||
build.queue_ttl = duration_pb2.Duration(seconds=timeout_in_seconds)
|
||||
|
||||
if tag:
|
||||
build.tags = [tag]
|
||||
|
||||
@@ -25,4 +25,4 @@ steps:
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/execute_notebook_cli.py --notebook_source "${_NOTEBOOK_GCS_URI}" --output_file_or_uri "${_NOTEBOOK_OUTPUT_GCS_URI}"'
|
||||
env:
|
||||
- 'IS_TESTING=1'
|
||||
timeout: 86400s
|
||||
timeout: 86400s
|
||||
@@ -5,22 +5,15 @@ steps:
|
||||
args:
|
||||
- -c
|
||||
- 'gcloud config list'
|
||||
# # Clone the Git repo
|
||||
# - name: ${_PYTHON_IMAGE}
|
||||
# entrypoint: git
|
||||
# args: ['clone', "${_GIT_REPO}", "--branch", "${_GIT_BRANCH_NAME}", "."]
|
||||
# Check the Python version
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
args:
|
||||
- -c
|
||||
- 'python3 .cloud-build/CheckPythonVersion.py'
|
||||
# Fetch base branch if required
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
args:
|
||||
- -c
|
||||
- 'if [ -n "${_BASE_BRANCH}" ]; then git fetch origin "${_BASE_BRANCH}":refs/remotes/origin/"${_BASE_BRANCH}"; else echo "Skipping fetch."; fi'
|
||||
# Fetch full repo for diff purposes
|
||||
- name: gcr.io/cloud-builders/git
|
||||
args: [fetch, --unshallow]
|
||||
# Install Python dependencies
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
@@ -28,11 +21,15 @@ steps:
|
||||
- -c
|
||||
- 'python3 -m pip install -U pip && python3 -m pip install -U --user -r .cloud-build/requirements.txt'
|
||||
# Install Python dependencies and run testing script
|
||||
# TODO: Only pass in private_pool_id if it is set
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
args:
|
||||
- -c
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/ExecuteChangedNotebooks.py --test_paths_file "${_TEST_PATHS_FILE}" --base_branch "${_FORCED_BASE_BRANCH}" --container_uri ${_PYTHON_IMAGE} --staging_bucket ${_GCS_STAGING_BUCKET} --artifacts_bucket ${_GCS_STAGING_BUCKET}/executed_notebooks/PR_${_PR_NUMBER}/BUILD_${BUILD_ID} --variable_project_id ${PROJECT_ID} --variable_region ${_GCP_REGION}'
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/execute_changed_notebooks_cli.py --test_paths_file "${_TEST_PATHS_FILE}" --base_branch "${_FORCED_BASE_BRANCH}" --container_uri ${_PYTHON_IMAGE} --staging_bucket ${_GCS_STAGING_BUCKET} --artifacts_bucket ${_GCS_STAGING_BUCKET}/executed_notebooks/PR_${_PR_NUMBER}/BUILD_${BUILD_ID} --variable_project_id ${PROJECT_ID} --variable_region ${_GCP_REGION} `if [ ! -z "${_PRIVATE_POOL_NAME}" ]; then echo "--private_pool_id ${_PRIVATE_POOL_NAME}"; fi`'
|
||||
env:
|
||||
- 'IS_TESTING=1'
|
||||
timeout: 86400s
|
||||
options:
|
||||
pool:
|
||||
name: ${_PRIVATE_POOL_NAME}
|
||||
@@ -1,12 +1,13 @@
|
||||
ipython==7.30.1
|
||||
jupyter==1.0.0
|
||||
nbconvert==6.3.0
|
||||
papermill==2.3.3
|
||||
numpy==1.21.4
|
||||
pandas==1.3.5
|
||||
matplotlib==3.5.1
|
||||
ipython
|
||||
numpy
|
||||
jupyter
|
||||
nbconvert
|
||||
papermill
|
||||
pandas
|
||||
matplotlib
|
||||
tabulate
|
||||
google-cloud-aiplatform
|
||||
google-cloud-storage
|
||||
google-cloud-build
|
||||
gcloud
|
||||
ratemate
|
||||
GitPython
|
||||
@@ -0,0 +1,5 @@
|
||||
notebooks/official/vizier/gapic-vizier-multi-objective-optimization.ipynb
|
||||
notebooks/official/pipelines/lightweight_functions_component_io_kfp.ipynb
|
||||
notebooks/official/matching_engine/intro-swivel.ipynb
|
||||
notebooks/official/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb
|
||||
notebooks/official/pipelines/metrics_viz_run_compare_kfp.ipynb
|
||||
@@ -13,8 +13,10 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from nbconvert.preprocessors import Preprocessor
|
||||
from typing import Dict
|
||||
|
||||
from nbconvert.preprocessors import Preprocessor
|
||||
|
||||
from . import UpdateNotebookVariables as update_notebook_variables
|
||||
|
||||
|
||||
@@ -60,4 +62,4 @@ class UpdateVariablesPreprocessor(Preprocessor):
|
||||
|
||||
executable_cells.append(cell)
|
||||
notebook.cells = executable_cells
|
||||
return notebook, resources
|
||||
return notebook, resources
|
||||
|
||||
@@ -78,4 +78,4 @@ def test_region():
|
||||
variable_name="REGION",
|
||||
variable_value="us-central1",
|
||||
)
|
||||
assert new_content == 'REGION = "us-central1" # @param {type:"string"}'
|
||||
assert new_content == 'REGION = "us-central1" # @param {type:"string"}'
|
||||
|
||||
@@ -1,17 +1,17 @@
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
from google.cloud import storage
|
||||
from google.cloud.aiplatform import utils
|
||||
from google.auth import credentials as auth_credentials
|
||||
import os
|
||||
|
||||
import subprocess
|
||||
import tarfile
|
||||
import uuid
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
|
||||
from google.auth import credentials as auth_credentials
|
||||
from google.cloud import storage
|
||||
from google.cloud.aiplatform import utils
|
||||
|
||||
|
||||
def download_file(bucket_name: str, blob_name: str, destination_file: str) -> str:
|
||||
"""Copies a remote GCS file to a local path."""
|
||||
"""Copies a remote GCS file to a local path"""
|
||||
remote_file_path = "".join(["gs://", "/".join([bucket_name, blob_name])])
|
||||
|
||||
subprocess.check_output(
|
||||
@@ -25,7 +25,7 @@ def upload_file(
|
||||
local_file_path: str,
|
||||
remote_file_path: str,
|
||||
) -> str:
|
||||
"""Copies a local file to a GCS path."""
|
||||
"""Copies a local file to a GCS path"""
|
||||
subprocess.check_output(
|
||||
["gsutil", "cp", local_file_path, remote_file_path], encoding="UTF-8"
|
||||
)
|
||||
@@ -57,4 +57,4 @@ def archive_code_and_upload(staging_bucket: str):
|
||||
|
||||
print(f"Uploaded source code archive to {source_archived_file_gcs}")
|
||||
|
||||
return source_archived_file_gcs
|
||||
return source_archived_file_gcs
|
||||
|
||||
@@ -2,17 +2,17 @@ If you are opening a PR for `Official Notebooks` under the [notebooks/official](
|
||||
- [ ] Use the [notebook template](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb) as a starting point.
|
||||
- [ ] Follow the style and grammar rules outlined in the above notebook template.
|
||||
- [ ] Verify the notebook runs successfully in Colab since the automated tests cannot guarantee this even when it passes.
|
||||
- [ ] Passes all the required automated checks. You can locally test for formatting and linting with these [instructions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/docs/contributing.md#code-quality-checks).
|
||||
- [ ] Passes all the required automated checks. You can locally test for formatting and linting with these [instructions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/CONTRIBUTING.md#code-quality-checks).
|
||||
- [ ] You have consulted with a tech writer to see if tech writer review is necessary. If so, the notebook has been reviewed by a tech writer, and they have approved it.
|
||||
- [ ] This notebook has been added to the [CODEOWNERS](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/docs/CODEOWNERS) file under `# Official Notebooks` section, pointing to the author or the author's team.
|
||||
- [ ] This notebook has been added to the [CODEOWNERS](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/CODEOWNERS) file under the `Official Notebooks` section, pointing to the author or the author's team.
|
||||
- [ ] The Jupyter notebook cleans up any artifacts it has created (datasets, ML models, endpoints, etc) so as not to eat up unnecessary resources.
|
||||
|
||||
|
||||
If you are opening a PR for `Community Notebooks` under the [notebooks/community](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/community) folder:
|
||||
- [ ] This notebook has been added to the [CODEOWNERS](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/docs/CODEOWNERS) file under the `# Community Notebooks` section, pointing to the author or the author's team.
|
||||
- [ ] Passes all the required formatting and linting checks. You can locally test with these [instructions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/docs/contributing.md#code-quality-checks).
|
||||
- [ ] This notebook has been added to the [CODEOWNERS](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/CODEOWNERS) file under the `Community Notebooks` section, pointing to the author or the author's team.
|
||||
- [ ] Passes all the required formatting and linting checks. You can locally test with these [instructions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/CONTRIBUTING.md#code-quality-checks).
|
||||
|
||||
If you are opening a PR for `Community Content` under the [community-content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/community-content) folder:
|
||||
- [ ] Make sure your main `Content Directory Name` is descriptive, informative, and includes some of the key products and attributes of your content, so that it is differentiable from other content
|
||||
- [ ] The main content directory has been added to the [CODEOWNERS](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/docs/CODEOWNERS) file under the `# Community Content` section, pointing to the author or the author's team.
|
||||
- [ ] Passes all the required formatting and linting checks. You can locally test with these [instructions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/docs/contributing.md#code-quality-checks).
|
||||
- [ ] The main content directory has been added to the [CODEOWNERS](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/community-content/CODEOWNERS) file under the `Community Content` section, pointing to the author or the author's team.
|
||||
- [ ] Passes all the required formatting and linting checks. You can locally test with these [instructions](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/CONTRIBUTING.md#code-quality-checks).
|
||||
|
||||
@@ -7,9 +7,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v2
|
||||
uses: actions/setup-python@v4
|
||||
with:
|
||||
python-version: '3.x'
|
||||
- name: Fetch pull request branch
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Fetch base main branch
|
||||
|
||||
@@ -2,8 +2,9 @@ git+https://github.com/tensorflow/docs
|
||||
ipython
|
||||
jupyter
|
||||
nbconvert
|
||||
black==21.10b0
|
||||
pyupgrade==2.29.1
|
||||
black==22.3.0
|
||||
pyupgrade==2.34.0
|
||||
isort==5.10.1
|
||||
flake8==4.0.1
|
||||
nbqa==1.2.2
|
||||
nbqa==1.3.1
|
||||
|
||||
|
||||
@@ -84,19 +84,19 @@ if [ ${#notebooks[@]} -gt 0 ]; then
|
||||
FLAKE8_RTN=$?
|
||||
else
|
||||
echo "Running black..."
|
||||
python3 -m nbqa black "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa black "$notebook"
|
||||
BLACK_RTN=$?
|
||||
echo "Running pyupgrade..."
|
||||
python3 -m nbqa pyupgrade "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa pyupgrade "$notebook"
|
||||
PYUPGRADE_RTN=$?
|
||||
echo "Running isort..."
|
||||
python3 -m nbqa isort "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa isort "$notebook"
|
||||
ISORT_RTN=$?
|
||||
echo "Running nbfmt..."
|
||||
python3 -m tensorflow_docs.tools.nbfmt --remove_outputs "$notebook"
|
||||
NBFMT_RTN=$?
|
||||
echo "Running flake8..."
|
||||
python3 -m nbqa flake8 "$notebook" --show-source --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291 --nbqa-mutate
|
||||
python3 -m nbqa flake8 "$notebook" --show-source --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291
|
||||
FLAKE8_RTN=$?
|
||||
fi
|
||||
|
||||
|
||||
@@ -31,7 +31,7 @@ pip3 install --user -U nbqa black flake8 isort pyupgrade git+https://github.com/
|
||||
You'll likely need to add the directory where these were installed to your PATH:
|
||||
|
||||
```shell
|
||||
export PATH=“$HOME/.local/bin:$PATH"
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
```
|
||||
|
||||
Then, set an environment variable for your notebook (or directory):
|
||||
@@ -48,8 +48,8 @@ then you will need to manually address them before submitting your PR.
|
||||
nbqa black "$notebook"
|
||||
nbqa pyupgrade "$notebook"
|
||||
nbqa isort "$notebook"
|
||||
python3 -m tensorflow_docs.tools.nbfmt --remove_outputs "$notebook"
|
||||
nbqa flake8 "$notebook" --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291
|
||||
python3 -m tensorflow_docs.tools.nbfmt --remove_outputs "$notebook"
|
||||
```
|
||||
|
||||
## Code Reviews
|
||||
|
||||
@@ -6,7 +6,19 @@ Welcome to the Google Cloud [Vertex AI](https://cloud.google.com/vertex-ai/docs/
|
||||
|
||||
## Overview
|
||||
|
||||
The repository contains [Notebooks](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks) and [Community Content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/community-content) that demonstrate how to develop and manage ML workflows using Google Cloud Vertex AI.
|
||||
The repository contains [notebooks](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks) and [community content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/community-content) that demonstrate how to develop and manage ML workflows using Google Cloud Vertex AI.
|
||||
|
||||
## Repository structure
|
||||
|
||||
```bash
|
||||
├── community-content - Sample code and tutorials contributed by the community
|
||||
├── notebooks
|
||||
│ ├── community - Notebooks contributed by the community
|
||||
│ ├── official - Notebooks demonstrating use of each Vertex AI service
|
||||
│ │ ├── automl
|
||||
│ │ ├── custom
|
||||
│ │ ├── ...
|
||||
```
|
||||
|
||||
## Contributing
|
||||
|
||||
@@ -19,3 +31,7 @@ Please use the [issues page](https://github.com/GoogleCloudPlatform/vertex-ai-sa
|
||||
## Disclaimer
|
||||
|
||||
This is not an officially supported Google product. The code in this repository is for demonstrative purposes only.
|
||||
|
||||
## Feedback
|
||||
|
||||
Please feel free to fill out our [survey](https://bit.ly/vertex-ai-samples-survey) to give us feedback on the repo and its content.
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
* @vertex-ai-samples-contributors @GoogleCloudPlatform/cloudml-samples-owners
|
||||
/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk @yinghsienwu
|
||||
/pytorch_text_classification_using_vertex_sdk_and_gcloud @RajeshThallam
|
||||
/pytorch_text_classification_using_vertex_sdk_and_gcloud @RajeshThallam @ultrons
|
||||
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
|
||||
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
|
||||
/pluto_on_workbench @wkharold
|
||||
|
||||
@@ -0,0 +1,824 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "pc5-mbsX9PZC"
|
||||
},
|
||||
"source": [
|
||||
"# AlphaFold On Vertex AI Workbench\n",
|
||||
"\n",
|
||||
"[Vertex AI Workbench](https://cloud.google.com/vertex-ai/docs/workbench) offers an end-to-end notebook-based production environment that can be preconfigured with the runtime dependencies necessary to run AlphaFold on Vertex AI. With [User-Managed Notebooks](https://cloud.google.com/vertex-ai/docs/workbench/user-managed/introduction), you can configure a GPU accelerator to run AlphaFold using Tensorflow, without having to install and manage drivers or JupyterLab instances. This notebook allows you to easily predict the structure of a protein using a slightly simplified version of [AlphaFold v2.1.0](https://doi.org/10.1038/s41586-021-03819-2). \n",
|
||||
"\n",
|
||||
"##  [Launch this Notebook in Vertex AI Workbench](https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/raw/main/community-content/alphafold_on_workbench/AlphaFold.ipynb)\n",
|
||||
"\n",
|
||||
"**Differences to AlphaFold v2.1.0**\n",
|
||||
"\n",
|
||||
"In comparison to AlphaFold v2.1.0, this notebook notebook uses **no templates (homologous structures)** and a selected portion of the [BFD database](https://bfd.mmseqs.com/). We have validated these changes on several thousand recent PDB structures. While accuracy will be near-identical to the full AlphaFold system on many targets, a small fraction have a large drop in accuracy due to the smaller MSA and lack of templates. For best reliability, we recommend instead using the [full open source AlphaFold](https://github.com/deepmind/alphafold/), or the [AlphaFold Protein Structure Database](https://alphafold.ebi.ac.uk/).\n",
|
||||
"\n",
|
||||
"**This notebook has an small drop in average accuracy for multimers compared to local AlphaFold installation, for full multimer accuracy it is highly recommended to run [AlphaFold locally](https://github.com/deepmind/alphafold#running-alphafold).** Moreover, the AlphaFold-Multimer requires searching for MSA for every unique sequence in the complex, hence it is substantially slower. If your notebook times-out due to slow multimer MSA search, we recommend running AlphaFold locally.\n",
|
||||
"\n",
|
||||
"Please note that this notebook is provided as an early-access prototype and is not a finished product. It is provided for theoretical modelling only and caution should be exercised in its use. \n",
|
||||
"\n",
|
||||
"**Citing this work**\n",
|
||||
"\n",
|
||||
"Any publication that discloses findings arising from using this notebook should [cite](https://github.com/deepmind/alphafold/#citing-this-work) the [AlphaFold paper](https://doi.org/10.1038/s41586-021-03819-2).\n",
|
||||
"\n",
|
||||
"**Licenses**\n",
|
||||
"\n",
|
||||
"This Colab uses the [AlphaFold model parameters](https://github.com/deepmind/alphafold/#model-parameters-license) which are subject to the Creative Commons Attribution 4.0 International ([CC BY 4.0](https://creativecommons.org/licenses/by/4.0/legalcode)) license. The Colab itself is provided under the [Apache 2.0 license](https://www.apache.org/licenses/LICENSE-2.0). See the full license statement below.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"**More information**\n",
|
||||
"\n",
|
||||
"You can find more information about how AlphaFold works in the following papers:\n",
|
||||
"\n",
|
||||
"* [AlphaFold methods paper](https://www.nature.com/articles/s41586-021-03819-2)\n",
|
||||
"* [AlphaFold predictions of the human proteome paper](https://www.nature.com/articles/s41586-021-03828-1)\n",
|
||||
"* [AlphaFold-Multimer paper](https://www.biorxiv.org/content/10.1101/2021.10.04.463034v1)\n",
|
||||
"\n",
|
||||
"FAQ on how to interpret AlphaFold predictions are [here](https://alphafold.ebi.ac.uk/faq)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "b7a02613eb1a"
|
||||
},
|
||||
"source": [
|
||||
"## Download AlphaFold Data"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "woIxeCPygt7K"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import subprocess\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"import alphafold.common\n",
|
||||
"import tqdm.notebook\n",
|
||||
"from IPython.utils import io\n",
|
||||
"\n",
|
||||
"TQDM_BAR_FORMAT = (\n",
|
||||
" \"{l_bar}{bar}| {n_fmt}/{total_fmt} [elapsed: {elapsed} remaining: {remaining}]\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"SOURCE_URL = (\n",
|
||||
" \"https://storage.googleapis.com/alphafold/alphafold_params_colab_2022-01-19.tar\"\n",
|
||||
")\n",
|
||||
"PARAMS_DIR = \"alphafold/data/params\"\n",
|
||||
"PARAMS_PATH = os.path.join(PARAMS_DIR, os.path.basename(SOURCE_URL))\n",
|
||||
"ALPHAFOLD_COMMON_DIR = os.path.dirname(alphafold.common.__file__)\n",
|
||||
"\n",
|
||||
"try:\n",
|
||||
" with tqdm.notebook.tqdm(total=100, bar_format=TQDM_BAR_FORMAT) as pbar:\n",
|
||||
" with io.capture_output() as captured:\n",
|
||||
"\n",
|
||||
" # Download and store stereo_chemical_props.txt\n",
|
||||
" !mkdir -p ~/content/alphafold/alphafold/common\n",
|
||||
" !mkdir -p /opt/conda/lib/python3.7/site-packages/alphafold/common/\n",
|
||||
" !wget -q -P ~/content/alphafold/alphafold/common https://git.scicore.unibas.ch/schwede/openstructure/-/raw/7102c63615b64735c4941278d92b554ec94415f8/modules/mol/alg/src/stereo_chemical_props.txt\n",
|
||||
" pbar.update(18)\n",
|
||||
" !cp -f ~/content/alphafold/alphafold/common/stereo_chemical_props.txt \"{ALPHAFOLD_COMMON_DIR}\"\n",
|
||||
"\n",
|
||||
" # Download alphafold_params_colab_2021-10-27.tar\n",
|
||||
" !mkdir --parents \"{PARAMS_DIR}\"\n",
|
||||
" !wget -O \"{PARAMS_PATH}\" \"{SOURCE_URL}\"\n",
|
||||
" pbar.update(27)\n",
|
||||
"\n",
|
||||
" # Un-tar alphafold_params_colab_2021-10-27.tar\n",
|
||||
" !tar --extract --verbose --file=\"{PARAMS_PATH}\" --directory=\"{PARAMS_DIR}\" --preserve-permissions\n",
|
||||
" # !rm \"{PARAMS_PATH}\"\n",
|
||||
" pbar.update(55)\n",
|
||||
"\n",
|
||||
"except subprocess.CalledProcessError:\n",
|
||||
" print(captured)\n",
|
||||
" raise"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d8926b7d5529"
|
||||
},
|
||||
"source": [
|
||||
"## Configure GPU Acceleration"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "VzJ5iMjTtoZw"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Confirm accelerator configuration\n",
|
||||
"import jax\n",
|
||||
"\n",
|
||||
"if jax.local_devices()[0].platform == \"tpu\":\n",
|
||||
" raise RuntimeError(\n",
|
||||
" \"TPU runtime not supported. Please configure GPU acceleration on the VM.\"\n",
|
||||
" )\n",
|
||||
"elif jax.local_devices()[0].platform == \"cpu\":\n",
|
||||
" print(\n",
|
||||
" \"CPU-only runtime is not recommended, because prediction execution will be slow. For better performance, consider GPU acceleration on the VM.\"\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" print(f\"Running with {jax.local_devices()[0].device_kind} GPU\")\n",
|
||||
"\n",
|
||||
"# Make sure all necessary environment variables are set.\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"os.environ[\"TF_FORCE_UNIFIED_MEMORY\"] = \"1\"\n",
|
||||
"os.environ[\"XLA_PYTHON_CLIENT_MEM_FRACTION\"] = \"2.0\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "W4JpOs6oA-QS"
|
||||
},
|
||||
"source": [
|
||||
"## Making a prediction\n",
|
||||
"\n",
|
||||
"Please paste the sequence of your protein in the text box below, then run the remaining cells via _Run_ > _Run Selected Cell and All Below_. You can also run the cells individually by pressing the _Play_ button on the left.\n",
|
||||
"\n",
|
||||
"Note that the search against databases and the actual prediction can take some time, from minutes to hours, depending on the length of the protein and what type of GPU you allocate (see FAQ below).\n",
|
||||
"\n",
|
||||
"To start, enter the amino acid sequence(s) to fold ⬇️\n",
|
||||
"\n",
|
||||
"If you enter only a single sequence, the monomer model will be used. If you enter multiple sequences, the multimer model will be used."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b310d44229d0"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Input sequences (type: str)\n",
|
||||
"sequence_1 = \"MAAHKGAEHHHKAAEHHEQAAKHHHAAAEHHEKGEHEQAAHHADTAYAHHKHAEEHAAQAAKHDAEHHAPKPH\"\n",
|
||||
"sequence_2 = \"\"\n",
|
||||
"sequence_3 = \"\"\n",
|
||||
"sequence_4 = \"\"\n",
|
||||
"sequence_5 = \"\"\n",
|
||||
"sequence_6 = \"\"\n",
|
||||
"sequence_7 = \"\"\n",
|
||||
"sequence_8 = \"\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "rowN0bVYLe9n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from alphafold.notebooks import notebook_utils\n",
|
||||
"\n",
|
||||
"input_sequences = (\n",
|
||||
" sequence_1,\n",
|
||||
" sequence_2,\n",
|
||||
" sequence_3,\n",
|
||||
" sequence_4,\n",
|
||||
" sequence_5,\n",
|
||||
" sequence_6,\n",
|
||||
" sequence_7,\n",
|
||||
" sequence_8,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# If folding a complex target and all the input sequences are\n",
|
||||
"# prokaryotic then set `is_prokaryotic` to `True`. Set to `False`\n",
|
||||
"# otherwise or if the origin is unknown.\n",
|
||||
"\n",
|
||||
"is_prokaryote = False # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"MIN_SINGLE_SEQUENCE_LENGTH = 16\n",
|
||||
"MAX_SINGLE_SEQUENCE_LENGTH = 2500\n",
|
||||
"MAX_MULTIMER_LENGTH = 2500\n",
|
||||
"\n",
|
||||
"# Validate the input.\n",
|
||||
"sequences, model_type_to_use = notebook_utils.validate_input(\n",
|
||||
" input_sequences=input_sequences,\n",
|
||||
" min_length=MIN_SINGLE_SEQUENCE_LENGTH,\n",
|
||||
" max_length=MAX_SINGLE_SEQUENCE_LENGTH,\n",
|
||||
" max_multimer_length=MAX_MULTIMER_LENGTH,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "db551d4877ea"
|
||||
},
|
||||
"source": [
|
||||
"## Search against genetic databases\n",
|
||||
"\n",
|
||||
"Once this cell has been executed, you will see statistics about the multiple sequence alignment (MSA) that will be used by AlphaFold. In particular, you’ll see how well each residue is covered by similar sequences in the MSA."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "2tTeTTsLKPjB"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import collections\n",
|
||||
"import copy\n",
|
||||
"import random\n",
|
||||
"from concurrent import futures\n",
|
||||
"from urllib import request\n",
|
||||
"\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import numpy as np\n",
|
||||
"import py3Dmol\n",
|
||||
"from alphafold.common import protein\n",
|
||||
"from alphafold.data import (feature_processing, msa_pairing, pipeline,\n",
|
||||
" pipeline_multimer)\n",
|
||||
"from alphafold.data.tools import jackhmmer\n",
|
||||
"from alphafold.model import config, data, model\n",
|
||||
"from alphafold.relax import relax, utils\n",
|
||||
"from IPython import display\n",
|
||||
"from ipywidgets import GridspecLayout, Output\n",
|
||||
"\n",
|
||||
"# Color bands for visualizing plddt\n",
|
||||
"PLDDT_BANDS = [\n",
|
||||
" (0, 50, \"#FF7D45\"),\n",
|
||||
" (50, 70, \"#FFDB13\"),\n",
|
||||
" (70, 90, \"#65CBF3\"),\n",
|
||||
" (90, 100, \"#0053D6\"),\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"# --- Find the closest source ---\n",
|
||||
"test_url_pattern = (\n",
|
||||
" \"https://storage.googleapis.com/alphafold-colab{:s}/latest/uniref90_2021_03.fasta.1\"\n",
|
||||
")\n",
|
||||
"ex = futures.ThreadPoolExecutor(3)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def fetch(source):\n",
|
||||
" request.urlretrieve(test_url_pattern.format(source))\n",
|
||||
" return source\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"fs = [ex.submit(fetch, source) for source in [\"\", \"-europe\", \"-asia\"]]\n",
|
||||
"source = None\n",
|
||||
"for f in futures.as_completed(fs):\n",
|
||||
" source = f.result()\n",
|
||||
" ex.shutdown()\n",
|
||||
" break\n",
|
||||
"\n",
|
||||
"JACKHMMER_BINARY_PATH = \"/usr/bin/jackhmmer\"\n",
|
||||
"DB_ROOT_PATH = f\"https://storage.googleapis.com/alphafold-colab{source}/latest/\"\n",
|
||||
"# The z_value is the number of sequences in a database.\n",
|
||||
"MSA_DATABASES = [\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"uniref90\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}uniref90_2021_03.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 59,\n",
|
||||
" \"z_value\": 135_301_051,\n",
|
||||
" },\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"smallbfd\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}bfd-first_non_consensus_sequences.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 17,\n",
|
||||
" \"z_value\": 65_984_053,\n",
|
||||
" },\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"mgnify\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}mgy_clusters_2019_05.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 71,\n",
|
||||
" \"z_value\": 304_820_129,\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"# Search UniProt and construct the all_seq features only for heteromers, not homomers.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MULTIMER and len(set(sequences)) > 1:\n",
|
||||
" MSA_DATABASES.extend(\n",
|
||||
" [\n",
|
||||
" # Swiss-Prot and TrEMBL are concatenated together as UniProt.\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"uniprot\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}uniprot_2021_03.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 98,\n",
|
||||
" \"z_value\": 219_174_961 + 565_254,\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"TOTAL_JACKHMMER_CHUNKS = sum(cfg[\"num_streamed_chunks\"] for cfg in MSA_DATABASES)\n",
|
||||
"\n",
|
||||
"MAX_HITS = {\n",
|
||||
" \"uniref90\": 10_000,\n",
|
||||
" \"smallbfd\": 5_000,\n",
|
||||
" \"mgnify\": 501,\n",
|
||||
" \"uniprot\": 50_000,\n",
|
||||
"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_msa(fasta_path):\n",
|
||||
" \"\"\"Searches for MSA for the given sequence using chunked Jackhmmer search.\"\"\"\n",
|
||||
"\n",
|
||||
" # Run the search against chunks of genetic databases.\n",
|
||||
" raw_msa_results = collections.defaultdict(list)\n",
|
||||
" with tqdm.notebook.tqdm(\n",
|
||||
" total=TOTAL_JACKHMMER_CHUNKS, bar_format=TQDM_BAR_FORMAT\n",
|
||||
" ) as pbar:\n",
|
||||
"\n",
|
||||
" def jackhmmer_chunk_callback(i):\n",
|
||||
" pbar.update(n=1)\n",
|
||||
"\n",
|
||||
" for db_config in MSA_DATABASES:\n",
|
||||
" db_name = db_config[\"db_name\"]\n",
|
||||
" pbar.set_description(f\"Searching {db_name}\")\n",
|
||||
" jackhmmer_runner = jackhmmer.Jackhmmer(\n",
|
||||
" binary_path=JACKHMMER_BINARY_PATH,\n",
|
||||
" database_path=db_config[\"db_path\"],\n",
|
||||
" get_tblout=True,\n",
|
||||
" num_streamed_chunks=db_config[\"num_streamed_chunks\"],\n",
|
||||
" streaming_callback=jackhmmer_chunk_callback,\n",
|
||||
" z_value=db_config[\"z_value\"],\n",
|
||||
" )\n",
|
||||
" # Group the results by database name.\n",
|
||||
" raw_msa_results[db_name].extend(jackhmmer_runner.query(fasta_path))\n",
|
||||
"\n",
|
||||
" return raw_msa_results\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"features_for_chain = {}\n",
|
||||
"raw_msa_results_for_sequence = {}\n",
|
||||
"for sequence_index, sequence in enumerate(sequences, start=1):\n",
|
||||
" print(f\"\\nGetting MSA for sequence {sequence_index}\")\n",
|
||||
"\n",
|
||||
" fasta_path = f\"target_{sequence_index}.fasta\"\n",
|
||||
" with open(fasta_path, \"wt\") as f:\n",
|
||||
" f.write(f\">query\\n{sequence}\")\n",
|
||||
"\n",
|
||||
" # Don't do redundant work for multiple copies of the same chain in the multimer.\n",
|
||||
" if sequence not in raw_msa_results_for_sequence:\n",
|
||||
" raw_msa_results = get_msa(fasta_path=fasta_path)\n",
|
||||
" raw_msa_results_for_sequence[sequence] = raw_msa_results\n",
|
||||
" else:\n",
|
||||
" raw_msa_results = copy.deepcopy(raw_msa_results_for_sequence[sequence])\n",
|
||||
"\n",
|
||||
" # Extract the MSAs from the Stockholm files.\n",
|
||||
" # NB: deduplication happens later in pipeline.make_msa_features.\n",
|
||||
" single_chain_msas = []\n",
|
||||
" uniprot_msa = None\n",
|
||||
" for db_name, db_results in raw_msa_results.items():\n",
|
||||
" merged_msa = notebook_utils.merge_chunked_msa(\n",
|
||||
" results=db_results, max_hits=MAX_HITS.get(db_name)\n",
|
||||
" )\n",
|
||||
" if merged_msa.sequences and db_name != \"uniprot\":\n",
|
||||
" single_chain_msas.append(merged_msa)\n",
|
||||
" msa_size = len(set(merged_msa.sequences))\n",
|
||||
" print(\n",
|
||||
" f\"{msa_size} unique sequences found in {db_name} for sequence {sequence_index}\"\n",
|
||||
" )\n",
|
||||
" elif merged_msa.sequences and db_name == \"uniprot\":\n",
|
||||
" uniprot_msa = merged_msa\n",
|
||||
"\n",
|
||||
" notebook_utils.show_msa_info(\n",
|
||||
" single_chain_msas=single_chain_msas, sequence_index=sequence_index\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Turn the raw data into model features.\n",
|
||||
" feature_dict = {}\n",
|
||||
" feature_dict.update(\n",
|
||||
" pipeline.make_sequence_features(\n",
|
||||
" sequence=sequence, description=\"query\", num_res=len(sequence)\n",
|
||||
" )\n",
|
||||
" )\n",
|
||||
" feature_dict.update(pipeline.make_msa_features(msas=single_chain_msas))\n",
|
||||
" # We don't use templates in AlphaFold notebook, add only empty placeholder features.\n",
|
||||
" feature_dict.update(\n",
|
||||
" notebook_utils.empty_placeholder_template_features(\n",
|
||||
" num_templates=0, num_res=len(sequence)\n",
|
||||
" )\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Construct the all_seq features only for heteromers, not homomers.\n",
|
||||
" if (\n",
|
||||
" model_type_to_use == notebook_utils.ModelType.MULTIMER\n",
|
||||
" and len(set(sequences)) > 1\n",
|
||||
" ):\n",
|
||||
" valid_feats = msa_pairing.MSA_FEATURES + (\n",
|
||||
" \"msa_uniprot_accession_identifiers\",\n",
|
||||
" \"msa_species_identifiers\",\n",
|
||||
" )\n",
|
||||
" all_seq_features = {\n",
|
||||
" f\"{k}_all_seq\": v\n",
|
||||
" for k, v in pipeline.make_msa_features([uniprot_msa]).items()\n",
|
||||
" if k in valid_feats\n",
|
||||
" }\n",
|
||||
" feature_dict.update(all_seq_features)\n",
|
||||
"\n",
|
||||
" features_for_chain[protein.PDB_CHAIN_IDS[sequence_index - 1]] = feature_dict\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Do further feature post-processing depending on the model type.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" np_example = features_for_chain[protein.PDB_CHAIN_IDS[0]]\n",
|
||||
"\n",
|
||||
"elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" all_chain_features = {}\n",
|
||||
" for chain_id, chain_features in features_for_chain.items():\n",
|
||||
" all_chain_features[chain_id] = pipeline_multimer.convert_monomer_features(\n",
|
||||
" chain_features, chain_id\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" all_chain_features = pipeline_multimer.add_assembly_features(all_chain_features)\n",
|
||||
"\n",
|
||||
" np_example = feature_processing.pair_and_merge(\n",
|
||||
" all_chain_features=all_chain_features, is_prokaryote=is_prokaryote\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Pad MSA to avoid zero-sized extra_msa.\n",
|
||||
" np_example = pipeline_multimer.pad_msa(np_example, min_num_seq=512)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "9640643486bd"
|
||||
},
|
||||
"source": [
|
||||
"## Run AlphaFold\n",
|
||||
"\n",
|
||||
"Once this cell has been executed, a zip-archive \"prediction.zip\" with the obtained prediction will be saved on the VM, and available for download to your computer in the sidebar. In case you are having issues with the relaxation stage, you can disable it below. Warning: This means that the prediction might have distracting small stereochemical violations."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "XUo6foMQxwS2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"run_relax = True\n",
|
||||
"\n",
|
||||
"# --- Run the model ---\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" model_names = config.MODEL_PRESETS[\"monomer\"] + (\"model_2_ptm\",)\n",
|
||||
"elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" model_names = config.MODEL_PRESETS[\"multimer\"]\n",
|
||||
"\n",
|
||||
"output_dir = \"prediction\"\n",
|
||||
"os.makedirs(output_dir, exist_ok=True)\n",
|
||||
"\n",
|
||||
"plddts = {}\n",
|
||||
"ranking_confidences = {}\n",
|
||||
"pae_outputs = {}\n",
|
||||
"unrelaxed_proteins = {}\n",
|
||||
"\n",
|
||||
"with tqdm.notebook.tqdm(total=len(model_names) + 1, bar_format=TQDM_BAR_FORMAT) as pbar:\n",
|
||||
" for model_name in model_names:\n",
|
||||
" pbar.set_description(f\"Running {model_name}\")\n",
|
||||
"\n",
|
||||
" cfg = config.model_config(model_name)\n",
|
||||
" if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" cfg.data.eval.num_ensemble = 1\n",
|
||||
" elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" cfg.model.num_ensemble_eval = 1\n",
|
||||
" params = data.get_model_haiku_params(model_name, \"./alphafold/data\")\n",
|
||||
" model_runner = model.RunModel(cfg, params)\n",
|
||||
" processed_feature_dict = model_runner.process_features(\n",
|
||||
" np_example, random_seed=0\n",
|
||||
" )\n",
|
||||
" prediction = model_runner.predict(\n",
|
||||
" processed_feature_dict, random_seed=random.randrange(sys.maxsize)\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" mean_plddt = prediction[\"plddt\"].mean()\n",
|
||||
"\n",
|
||||
" if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" if \"predicted_aligned_error\" in prediction:\n",
|
||||
" pae_outputs[model_name] = (\n",
|
||||
" prediction[\"predicted_aligned_error\"],\n",
|
||||
" prediction[\"max_predicted_aligned_error\"],\n",
|
||||
" )\n",
|
||||
" else:\n",
|
||||
" # Monomer models are sorted by mean pLDDT. Do not put monomer pTM models here as they\n",
|
||||
" # should never get selected.\n",
|
||||
" ranking_confidences[model_name] = prediction[\"ranking_confidence\"]\n",
|
||||
" plddts[model_name] = prediction[\"plddt\"]\n",
|
||||
" elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" # Multimer models are sorted by pTM+ipTM.\n",
|
||||
" ranking_confidences[model_name] = prediction[\"ranking_confidence\"]\n",
|
||||
" plddts[model_name] = prediction[\"plddt\"]\n",
|
||||
" pae_outputs[model_name] = (\n",
|
||||
" prediction[\"predicted_aligned_error\"],\n",
|
||||
" prediction[\"max_predicted_aligned_error\"],\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Set the b-factors to the per-residue plddt.\n",
|
||||
" final_atom_mask = prediction[\"structure_module\"][\"final_atom_mask\"]\n",
|
||||
" b_factors = prediction[\"plddt\"][:, None] * final_atom_mask\n",
|
||||
" unrelaxed_protein = protein.from_prediction(\n",
|
||||
" processed_feature_dict,\n",
|
||||
" prediction,\n",
|
||||
" b_factors=b_factors,\n",
|
||||
" remove_leading_feature_dimension=(\n",
|
||||
" model_type_to_use == notebook_utils.ModelType.MONOMER\n",
|
||||
" ),\n",
|
||||
" )\n",
|
||||
" unrelaxed_proteins[model_name] = unrelaxed_protein\n",
|
||||
"\n",
|
||||
" # Delete unused outputs to save memory.\n",
|
||||
" del model_runner\n",
|
||||
" del params\n",
|
||||
" del prediction\n",
|
||||
" pbar.update(n=1)\n",
|
||||
"\n",
|
||||
" # --- AMBER relax the best model ---\n",
|
||||
"\n",
|
||||
" # Find the best model according to the mean pLDDT.\n",
|
||||
" best_model_name = max(\n",
|
||||
" ranking_confidences.keys(), key=lambda x: ranking_confidences[x]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" if run_relax:\n",
|
||||
" pbar.set_description(\"AMBER relaxation\")\n",
|
||||
" amber_relaxer = relax.AmberRelaxation(\n",
|
||||
" max_iterations=0,\n",
|
||||
" tolerance=2.39,\n",
|
||||
" stiffness=10.0,\n",
|
||||
" exclude_residues=[],\n",
|
||||
" max_outer_iterations=3,\n",
|
||||
" )\n",
|
||||
" relaxed_pdb, _, _ = amber_relaxer.process(\n",
|
||||
" prot=unrelaxed_proteins[best_model_name]\n",
|
||||
" )\n",
|
||||
" else:\n",
|
||||
" print(\"Warning: Running without the relaxation stage.\")\n",
|
||||
" relaxed_pdb = protein.to_pdb(unrelaxed_proteins[best_model_name])\n",
|
||||
" pbar.update(n=1) # Finished AMBER relax.\n",
|
||||
"\n",
|
||||
"# Construct multiclass b-factors to indicate confidence bands\n",
|
||||
"# 0=very low, 1=low, 2=confident, 3=very high\n",
|
||||
"banded_b_factors = []\n",
|
||||
"for plddt in plddts[best_model_name]:\n",
|
||||
" for idx, (min_val, max_val, _) in enumerate(PLDDT_BANDS):\n",
|
||||
" if plddt >= min_val and plddt <= max_val:\n",
|
||||
" banded_b_factors.append(idx)\n",
|
||||
" break\n",
|
||||
"banded_b_factors = np.array(banded_b_factors)[:, None] * final_atom_mask\n",
|
||||
"to_visualize_pdb = utils.overwrite_b_factors(relaxed_pdb, banded_b_factors)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Write out the prediction\n",
|
||||
"pred_output_path = os.path.join(output_dir, \"selected_prediction.pdb\")\n",
|
||||
"with open(pred_output_path, \"w\") as f:\n",
|
||||
" f.write(relaxed_pdb)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# --- Visualise the prediction & confidence ---\n",
|
||||
"show_sidechains = True\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def plot_plddt_legend():\n",
|
||||
" \"\"\"Plots the legend for pLDDT.\"\"\"\n",
|
||||
" thresh = [\n",
|
||||
" \"Very low (pLDDT < 50)\",\n",
|
||||
" \"Low (70 > pLDDT > 50)\",\n",
|
||||
" \"Confident (90 > pLDDT > 70)\",\n",
|
||||
" \"Very high (pLDDT > 90)\",\n",
|
||||
" ]\n",
|
||||
"\n",
|
||||
" colors = [x[2] for x in PLDDT_BANDS]\n",
|
||||
"\n",
|
||||
" plt.figure(figsize=(2, 2))\n",
|
||||
" for c in colors:\n",
|
||||
" plt.bar(0, 0, color=c)\n",
|
||||
" plt.legend(thresh, frameon=False, loc=\"center\", fontsize=20)\n",
|
||||
" plt.xticks([])\n",
|
||||
" plt.yticks([])\n",
|
||||
" ax = plt.gca()\n",
|
||||
" ax.spines[\"right\"].set_visible(False)\n",
|
||||
" ax.spines[\"top\"].set_visible(False)\n",
|
||||
" ax.spines[\"left\"].set_visible(False)\n",
|
||||
" ax.spines[\"bottom\"].set_visible(False)\n",
|
||||
" plt.title(\"Model Confidence\", fontsize=20, pad=20)\n",
|
||||
" return plt\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Show the structure coloured by chain if the multimer model has been used.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" multichain_view = py3Dmol.view(width=800, height=600)\n",
|
||||
" multichain_view.addModelsAsFrames(to_visualize_pdb)\n",
|
||||
" multichain_style = {\"cartoon\": {\"colorscheme\": \"chain\"}}\n",
|
||||
" multichain_view.setStyle({\"model\": -1}, multichain_style)\n",
|
||||
" multichain_view.zoomTo()\n",
|
||||
" multichain_view.show()\n",
|
||||
"\n",
|
||||
"# Color the structure by per-residue pLDDT\n",
|
||||
"color_map = {i: bands[2] for i, bands in enumerate(PLDDT_BANDS)}\n",
|
||||
"view = py3Dmol.view(width=800, height=600)\n",
|
||||
"view.addModelsAsFrames(to_visualize_pdb)\n",
|
||||
"style = {\"cartoon\": {\"colorscheme\": {\"prop\": \"b\", \"map\": color_map}}}\n",
|
||||
"if show_sidechains:\n",
|
||||
" style[\"stick\"] = {}\n",
|
||||
"view.setStyle({\"model\": -1}, style)\n",
|
||||
"view.zoomTo()\n",
|
||||
"\n",
|
||||
"grid = GridspecLayout(1, 2)\n",
|
||||
"out = Output()\n",
|
||||
"with out:\n",
|
||||
" view.show()\n",
|
||||
"grid[0, 0] = out\n",
|
||||
"\n",
|
||||
"out = Output()\n",
|
||||
"with out:\n",
|
||||
" plot_plddt_legend().show()\n",
|
||||
"grid[0, 1] = out\n",
|
||||
"\n",
|
||||
"display.display(grid)\n",
|
||||
"\n",
|
||||
"# Display pLDDT and predicted aligned error (if output by the model).\n",
|
||||
"if pae_outputs:\n",
|
||||
" num_plots = 2\n",
|
||||
"else:\n",
|
||||
" num_plots = 1\n",
|
||||
"\n",
|
||||
"plt.figure(figsize=[8 * num_plots, 6])\n",
|
||||
"plt.subplot(1, num_plots, 1)\n",
|
||||
"plt.plot(plddts[best_model_name])\n",
|
||||
"plt.title(\"Predicted LDDT\")\n",
|
||||
"plt.xlabel(\"Residue\")\n",
|
||||
"plt.ylabel(\"pLDDT\")\n",
|
||||
"\n",
|
||||
"if num_plots == 2:\n",
|
||||
" plt.subplot(1, 2, 2)\n",
|
||||
" pae, max_pae = list(pae_outputs.values())[0]\n",
|
||||
" plt.imshow(pae, vmin=0.0, vmax=max_pae, cmap=\"Greens_r\")\n",
|
||||
" plt.colorbar(fraction=0.046, pad=0.04)\n",
|
||||
"\n",
|
||||
" # Display lines at chain boundaries.\n",
|
||||
" best_unrelaxed_prot = unrelaxed_proteins[best_model_name]\n",
|
||||
" total_num_res = best_unrelaxed_prot.residue_index.shape[-1]\n",
|
||||
" chain_ids = best_unrelaxed_prot.chain_index\n",
|
||||
" for chain_boundary in np.nonzero(chain_ids[:-1] - chain_ids[1:]):\n",
|
||||
" if chain_boundary.size:\n",
|
||||
" plt.plot([0, total_num_res], [chain_boundary, chain_boundary], color=\"red\")\n",
|
||||
" plt.plot([chain_boundary, chain_boundary], [0, total_num_res], color=\"red\")\n",
|
||||
"\n",
|
||||
" plt.title(\"Predicted Aligned Error\")\n",
|
||||
" plt.xlabel(\"Scored residue\")\n",
|
||||
" plt.ylabel(\"Aligned residue\")\n",
|
||||
"\n",
|
||||
"# Save the predicted aligned error (if it exists).\n",
|
||||
"pae_output_path = os.path.join(output_dir, \"predicted_aligned_error.json\")\n",
|
||||
"if pae_outputs:\n",
|
||||
" # Save predicted aligned error in the same format as the AF EMBL DB.\n",
|
||||
" pae_data = notebook_utils.get_pae_json(pae=pae, max_pae=max_pae.item())\n",
|
||||
" with open(pae_output_path, \"w\") as f:\n",
|
||||
" f.write(pae_data)\n",
|
||||
"\n",
|
||||
"!zip -q -r {output_dir}.zip {output_dir}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "lUQAn5LYC5n4"
|
||||
},
|
||||
"source": [
|
||||
"### Interpreting the prediction\n",
|
||||
"\n",
|
||||
"In general predicted LDDT (pLDDT) is best used for intra-domain confidence, whereas Predicted Aligned Error (PAE) is best used for determining between domain or between chain confidence.\n",
|
||||
"\n",
|
||||
"Please see the [AlphaFold methods paper](https://www.nature.com/articles/s41586-021-03819-2), the [AlphaFold predictions of the human proteome paper](https://www.nature.com/articles/s41586-021-03828-1), and the [AlphaFold-Multimer paper](https://www.biorxiv.org/content/10.1101/2021.10.04.463034v1) as well as [our FAQ](https://alphafold.ebi.ac.uk/faq) on how to interpret AlphaFold predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "jeb2z8DIA4om"
|
||||
},
|
||||
"source": [
|
||||
"## FAQ & Troubleshooting\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"* How do I get a predicted protein structure for my protein?\n",
|
||||
" * Connect the notebook to the Jupyter kernel \"Python 3 (ipykernel)\".\n",
|
||||
" * Paste the amino acid sequence of your protein (without any headers) into the variable sequence_1 in \"Making a Prediction\".\n",
|
||||
" * Run all cells in the notebook, either by running them individually or via \"Kernel\"/\"Restart Kernel and Run All Cells...\"\n",
|
||||
" * The predicted protein structure will be downloaded once all cells have been executed. Note: This can take minutes to hours - see below.\n",
|
||||
"* How long will this take?\n",
|
||||
" * The search against genetic databases can take minutes to hours.\n",
|
||||
" * Running AlphaFold and generating the prediction can take minutes to hours, depending on the length of your protein and on which GPU-type your VM has access to.\n",
|
||||
"* My notebook no longer seems to be doing anything, what should I do?\n",
|
||||
" * Some steps may take minutes to hours to complete.\n",
|
||||
" * If nothing happens or if you receive an error message, try restarting your notebook runtime via \"Kernel\"/\"Restart Kernel and Run All Cells...\".\n",
|
||||
" * If this doesn’t help, try resetting restarting your VM inside the GCloud Console (\"Compute Engine\"/\"VM Instances\").\n",
|
||||
"* How does this compare to the open-source version of AlphaFold?\n",
|
||||
" * This notebook version of AlphaFold searches a selected portion of the BFD dataset and currently doesn’t use templates, so its accuracy is reduced in comparison to the full version of AlphaFold that is described in the [AlphaFold paper](https://doi.org/10.1038/s41586-021-03819-2) and [Github repo](https://github.com/deepmind/alphafold/) (the full version is available via the inference script).\n",
|
||||
"* I received a warning “Notebook requires high RAM”, what do I do?\n",
|
||||
" * In the \"Compute Engine\"/\"VM Instances\" Console menu, you can reconfigure the host VM settings. See [Changing the machine type of a VM instance](https://cloud.google.com/compute/docs/instances/changing-machine-type-of-stopped-instance) for instructions.\n",
|
||||
"* Does this tool install anything on my computer?\n",
|
||||
" * No, everything happens in the VM instance within your Google Cloud project.\n",
|
||||
"* How should I share feedback and bug reports?\n",
|
||||
" * Please share any feedback and bug reports as an [issue](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues) on Github.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Related work\n",
|
||||
"\n",
|
||||
"Take a look at these Colab notebooks provided by the community (please note that these notebooks may vary from our validated AlphaFold system and we cannot guarantee their accuracy):\n",
|
||||
"\n",
|
||||
"* The [ColabFold AlphaFold2 notebook](https://colab.research.google.com/github/sokrypton/ColabFold/blob/main/AlphaFold2.ipynb) by Sergey Ovchinnikov, Milot Mirdita and Martin Steinegger, which uses an API hosted at the Södinglab based on the MMseqs2 server ([Mirdita et al. 2019, Bioinformatics](https://academic.oup.com/bioinformatics/article/35/16/2856/5280135)) for the multiple sequence alignment creation.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "YfPhvYgKC81B"
|
||||
},
|
||||
"source": [
|
||||
"# License and Disclaimer\n",
|
||||
"\n",
|
||||
"This is not an officially-supported Google product.\n",
|
||||
"\n",
|
||||
"This notebook and other information provided is for theoretical modelling only, caution should be exercised in its use. It is provided ‘as-is’ without any warranty of any kind, whether expressed or implied. Information is not intended to be a substitute for professional medical advice, diagnosis, or treatment, and does not constitute medical or other professional advice.\n",
|
||||
"\n",
|
||||
"Copyright 2021 DeepMind Technologies Limited.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## AlphaFold Code License\n",
|
||||
"\n",
|
||||
"Licensed under the Apache License, Version 2.0 (the \"License\"); you may not use this file except in compliance with the License. You may obtain a copy of the License at https://www.apache.org/licenses/LICENSE-2.0.\n",
|
||||
"\n",
|
||||
"Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on an \"AS IS\" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the specific language governing permissions and limitations under the License.\n",
|
||||
"\n",
|
||||
"## Model Parameters License\n",
|
||||
"\n",
|
||||
"The AlphaFold parameters are made available under the terms of the Creative Commons Attribution 4.0 International (CC BY 4.0) license. You can find details at: https://creativecommons.org/licenses/by/4.0/legalcode\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Third-party software\n",
|
||||
"\n",
|
||||
"Use of the third-party software, libraries or code referred to in the [Acknowledgements section](https://github.com/deepmind/alphafold/#acknowledgements) in the AlphaFold README may be governed by separate terms and conditions or license provisions. Your use of the third-party software, libraries or code is subject to any such terms and you should check that you can comply with any applicable restrictions or terms and conditions before use.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Mirrored Databases\n",
|
||||
"\n",
|
||||
"The following databases have been mirrored by DeepMind, and are available with reference to the following:\n",
|
||||
"* UniProt: v2021\\_03 (unmodified), by The UniProt Consortium, available under a [Creative Commons Attribution-NoDerivatives 4.0 International License](http://creativecommons.org/licenses/by-nd/4.0/).\n",
|
||||
"* UniRef90: v2021\\_03 (unmodified), by The UniProt Consortium, available under a [Creative Commons Attribution-NoDerivatives 4.0 International License](http://creativecommons.org/licenses/by-nd/4.0/).\n",
|
||||
"* MGnify: v2019\\_05 (unmodified), by Mitchell AL et al., available free of all copyright restrictions and made fully and freely available for both non-commercial and commercial use under [CC0 1.0 Universal (CC0 1.0) Public Domain Dedication](https://creativecommons.org/publicdomain/zero/1.0/).\n",
|
||||
"* BFD: (modified), by Steinegger M. and Söding J., modified by DeepMind, available under a [Creative Commons Attribution-ShareAlike 4.0 International License](https://creativecommons.org/licenses/by/4.0/). See the Methods section of the [AlphaFold proteome paper](https://www.nature.com/articles/s41586-021-03828-1) for details."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"accelerator": "GPU",
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "AlphaFold.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
# Copyright 2022 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
ARG CUDA_MAJOR=11
|
||||
ARG CUDA_MINOR=0
|
||||
|
||||
FROM gcr.io/deeplearning-platform-release/base-cu110
|
||||
|
||||
ARG CUDA_MAJOR
|
||||
ARG CUDA_MINOR
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
|
||||
build-essential \
|
||||
cmake \
|
||||
cuda-command-line-tools-${CUDA_MAJOR}-${CUDA_MINOR} \
|
||||
git \
|
||||
hmmer \
|
||||
kalign \
|
||||
tzdata \
|
||||
wget \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Compile HHsuite from source.
|
||||
RUN git clone --branch v3.3.0 https://github.com/soedinglab/hh-suite.git /tmp/hh-suite \
|
||||
&& mkdir /tmp/hh-suite/build \
|
||||
&& pushd /tmp/hh-suite/build \
|
||||
&& cmake -DCMAKE_INSTALL_PREFIX=/opt/hhsuite .. \
|
||||
&& make -j 4 && make install \
|
||||
&& ln -s /opt/hhsuite/bin/* /usr/bin \
|
||||
&& popd \
|
||||
&& rm -rf /tmp/hh-suite
|
||||
|
||||
ENV PATH="/opt/conda/bin:$PATH"
|
||||
RUN conda update -qy conda \
|
||||
&& conda install -y -c conda-forge \
|
||||
openmm=7.5.1 \
|
||||
cudatoolkit==${CUDA_VERSION} \
|
||||
pdbfixer \
|
||||
pip \
|
||||
python=3.7
|
||||
|
||||
COPY . /app/alphafold
|
||||
|
||||
# Install pip packages.
|
||||
RUN pip3 install --upgrade pip \
|
||||
&& pip3 install -r /app/alphafold/requirements.txt \
|
||||
&& pip3 install py3Dmol tqdm \
|
||||
&& pip3 install --upgrade jax==0.2.14 jaxlib==0.1.69+cuda${CUDA_MAJOR}${CUDA_MINOR} -f \
|
||||
https://storage.googleapis.com/jax-releases/jax_releases.html
|
||||
|
||||
# Install alphafold.
|
||||
WORKDIR /app/alphafold
|
||||
RUN python setup.py install
|
||||
|
||||
# Apply OpenMM patch.
|
||||
WORKDIR /opt/conda/lib/python3.7/site-packages
|
||||
RUN patch -p0 < /app/alphafold/docker/openmm.patch
|
||||
|
||||
# Creating a tmp location for jackhmmr; not mounting through to host though.
|
||||
RUN sudo mkdir -m 777 --parents /tmp/ramdisk
|
||||
|
||||
# We need to run `ldconfig` first to ensure GPUs are visible, due to some quirk
|
||||
# with Debian. See https://github.com/NVIDIA/nvidia-docker/issues/1399 for
|
||||
# details.
|
||||
# ENTRYPOINT does not support easily running multiple commands, so instead we
|
||||
# write a shell script to wrap them up.
|
||||
WORKDIR /home/jupyter
|
||||
RUN echo '#!/bin/bash\nldconfig\n\'
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
#!/usr/bin/env bash
|
||||
set -e
|
||||
|
||||
# Prod (Publicly viewable)
|
||||
PROJECT=cloud-devrel-public-resources
|
||||
REPOSITORY=alphafold
|
||||
LOCAL_IMAGE=alphafold-on-gcp
|
||||
REMOTE_IMAGE=${LOCAL_IMAGE?}
|
||||
TAG=latest
|
||||
REGISTRY="us-west1-docker.pkg.dev/${PROJECT?}/${REPOSITORY?}/${REMOTE_IMAGE?}:${TAG?}"
|
||||
|
||||
git clone https://github.com/deepmind/alphafold.git
|
||||
|
||||
cp Dockerfile alphafold/docker/Dockerfile
|
||||
cp AlphaFold.ipynb alphafold/notebooks/AlphaFold.ipynb
|
||||
|
||||
cd alphafold && sudo docker build --tag ${LOCAL_IMAGE?}:${TAG?} -f docker/Dockerfile .
|
||||
|
||||
sudo docker tag ${LOCAL_IMAGE?}:${TAG?} ${REGISTRY?}
|
||||
sudo docker push ${REGISTRY?}
|
||||
|
After Width: | Height: | Size: 3.1 KiB |
@@ -0,0 +1,52 @@
|
||||
# Overview
|
||||
*Pluto* is a programming environment for Julia, designed to be interactive and helpful. It provides a familiar notebook interface but it is not a Jupyter notebook. The biggest difference is that Pluto notebooks are reactive, changing a variable or function in one cell causes the cells that depend on that variable or function to be reevaluated. Pluto also provides useful interaction mechanisms that allow users to dynamically interact with the notebooks computation state.
|
||||
|
||||
The JuliaCon 2020 presentation: [Interactive notebooks ~ Pluto.jl]() provides a good introduction to Pluto. The source is at [fonsp/Pluto.jl]()
|
||||
|
||||
# Install Pluto
|
||||
|
||||
## Create a Vertex AI JupyterLab Instance
|
||||
|
||||
1. From the [GCP console](https://console.cloud.google.com) "hamburger menu"
|
||||
|
||||
select Vertex AI > Workbench
|
||||
2. Click NEW NOTEBOOK
|
||||
|
||||
* Choose Python 3 if you won't be using a GPU
|
||||
* Choose Python 3 (CUDA Toolkit xx.y) if you do want use a GPU
|
||||
3. Give the notebook an appropriate name
|
||||
4. Edit Notebook properties if you have special requirements otherwise accept the defaults and click CREATE
|
||||
5. When the notebook instance is ready click OPEN JUPYTERLAB
|
||||
|
||||
## Configure JupyterLab
|
||||
|
||||
1. Open a terminal by clicking the Terminal icon.
|
||||
1. Install the plutoserver
|
||||
pip3 install git+https://github.com/fonsp/pluto-on-jupyterlab.git
|
||||
1. In a browser go to [julialang.org/downloads](https://julialang.org/downloads/)
|
||||
1. In the Current stable release right click on the `Generic Linux on x86 / 64-bit (glibc)` link
|
||||
Select copy link address
|
||||
1. Back in the terminal switch to root via
|
||||
sudo -i
|
||||
1. Download the release to /opt and install julia in /usr/local/bin
|
||||
```bash
|
||||
cd /opt
|
||||
wget <paste the release link address>
|
||||
tar xf <name of the downloaded tar file>
|
||||
ln -s /opt/<julia-x.y.z>/bin/julia /usr/local/bin
|
||||
^d
|
||||
```
|
||||
1. Add the Pluto package to Julia
|
||||
```bash
|
||||
julia
|
||||
julia> ]add Pluto
|
||||
julia> bksp
|
||||
julia> using Pluto
|
||||
julia> ^d
|
||||
```
|
||||
1. From the JupyterLab menu bar select File > Shut Down
|
||||
|
||||
# Start Pluto
|
||||
1. Click OPEN JUPYTERLAB in the Workbench
|
||||
1. In the Notebook section of the Launcher click Pluto.jl
|
||||
1. The welcome to Pluto.jl screen should appear
|
||||
@@ -1,6 +1,6 @@
|
||||
# PyTorch on Google Cloud: Text Classification
|
||||
|
||||
In the PyTorch on Google Cloud series of blog posts, we aim to share how to build, train and deploy PyTorch models at scale and how to create reproducible machine learning pipelines on Google Cloud with [Vertex AI](https://cloud.google.com/vertex-ai).
|
||||
In the PyTorch on Google Cloud series of blog posts, we aim to share how to build, train, deploy and orchestrate PyTorch models at scale and how to create reproducible machine learning pipelines on Google Cloud with [Vertex AI](https://cloud.google.com/vertex-ai).
|
||||
|
||||
This tutorial on text classification shows how to train a PyTorch based text classification model by fine tuning a pre-trained Huggingface Transformers model and deploy the model on [Vertex AI](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) using Vertex SDK and [`gcloud ai`](https://cloud.google.com/sdk/gcloud/reference/beta/ai).
|
||||
|
||||
@@ -9,6 +9,7 @@ This tutorial on text classification shows how to train a PyTorch based text cla
|
||||
| <h4>Notebook</h4> | <h4>Description</h4> |
|
||||
| :-------- | :------- |
|
||||
| [pytorch-text-classification-vertex-ai-train-tune-deploy.ipynb](./pytorch-text-classification-vertex-ai-train-tune-deploy.ipynb) | Notebook to show training, hyper-parameter tuning and deploying a PyTorch model on Vertex AI |
|
||||
| [pytorch-text-classification-vertex-ai-pipelines.ipynb](./pytorch-text-classification-vertex-ai-pipelines.ipynb) | Notebook to show orchestration of PyTorch ML workflows on Vertex AI Pipelines using Kubeflow Pipelines SDK |
|
||||
|
||||
## Folders
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
|
||||
# Use pytorch GPU base image
|
||||
FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7
|
||||
# FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7
|
||||
FROM us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-10:latest
|
||||
|
||||
# set working directory
|
||||
WORKDIR /app
|
||||
|
||||
@@ -22,15 +22,18 @@ PROJECT_ID=$(gcloud config list --format 'value(core.project)')
|
||||
|
||||
# BUCKET_NAME: Change to your bucket name.
|
||||
BUCKET_NAME="[your-bucket-name]" # <-- CHANGE TO YOUR BUCKET NAME
|
||||
BUCKET_NAME=cloud-ai-platform-2f444b6a-a742-444b-b91a-c7519f51bd77
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# JOB_NAME: the name of your job running on AI Platform.
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-cstm-cntr-"
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-cstm-cntr"
|
||||
JOB_NAME=${JOB_PREFIX}-$(date +%Y%m%d%H%M%S)-custom-job
|
||||
|
||||
# This can be a GCS location to a zipped and uploaded package
|
||||
PACKAGE_PATH=./trainer
|
||||
|
||||
# REGION: select a region from https://cloud.google.com/vertex-ai/docs/general/locations#available_regions
|
||||
# or use the default '`us-central1`'. The region is where the job will be run.
|
||||
REGION="us-central1"
|
||||
@@ -41,11 +44,8 @@ JOB_DIR=gs://${BUCKET_NAME}/${JOB_PREFIX}/models/${JOB_NAME}
|
||||
# IMAGE_REPO_NAME: set a local repo name to distinquish our image
|
||||
IMAGE_REPO_NAME=pytorch_gpu_train_finetuned-bert-classifier
|
||||
|
||||
# IMAGE_TAG: an easily identifiable tag for your docker image
|
||||
IMAGE_TAG=latest
|
||||
|
||||
# IMAGE_URI: the complete URI location for Cloud Container Registry
|
||||
CUSTOM_TRAIN_IMAGE_URI=gcr.io/${PROJECT_ID}/${IMAGE_REPO_NAME}:${IMAGE_TAG}
|
||||
CUSTOM_TRAIN_IMAGE_URI=gcr.io/${PROJECT_ID}/${IMAGE_REPO_NAME}
|
||||
|
||||
# Build the docker image
|
||||
docker build --no-cache -f Dockerfile -t $CUSTOM_TRAIN_IMAGE_URI ../python_package
|
||||
@@ -53,11 +53,19 @@ docker build --no-cache -f Dockerfile -t $CUSTOM_TRAIN_IMAGE_URI ../python_packa
|
||||
# Deploy the docker image to Cloud Container Registry
|
||||
docker push ${CUSTOM_TRAIN_IMAGE_URI}
|
||||
|
||||
# worker pool spec
|
||||
worker_pool_spec="\
|
||||
replica-count=1,\
|
||||
machine-type=n1-standard-8,\
|
||||
accelerator-type=NVIDIA_TESLA_V100,\
|
||||
accelerator-count=1,\
|
||||
container-image-uri=${CUSTOM_TRAIN_IMAGE_URI}"
|
||||
|
||||
# Submit Custom Job to Vertex AI
|
||||
gcloud beta ai custom-jobs create \
|
||||
--display-name=${JOB_NAME} \
|
||||
--region ${REGION} \
|
||||
--worker-pool-spec=replica-count=1,machine-type='n1-standard-8',accelerator-type='NVIDIA_TESLA_V100',accelerator-count=1,container-image-uri=${CUSTOM_TRAIN_IMAGE_URI} \
|
||||
--worker-pool-spec="${worker_pool_spec}" \
|
||||
--args="--model-name","finetuned-bert-classifier","--job-dir",$JOB_DIR
|
||||
|
||||
echo "After the job is completed successfully, model files will be saved at $JOB_DIR/"
|
||||
|
||||
|
After Width: | Height: | Size: 45 KiB |
|
After Width: | Height: | Size: 37 KiB |
|
After Width: | Height: | Size: 76 KiB |
|
After Width: | Height: | Size: 74 KiB |
|
After Width: | Height: | Size: 248 KiB |
|
After Width: | Height: | Size: 38 KiB |
|
After Width: | Height: | Size: 123 KiB |
@@ -2,10 +2,13 @@
|
||||
FROM pytorch/torchserve:latest-cpu
|
||||
|
||||
# install dependencies
|
||||
RUN python3 -m pip install --upgrade pip
|
||||
RUN pip3 install transformers
|
||||
|
||||
USER model-server
|
||||
|
||||
# copy model artifacts, custom handler and other dependencies
|
||||
COPY ./custom_text_handler.py /home/model-server/
|
||||
COPY ./custom_handler.py /home/model-server/
|
||||
COPY ./index_to_name.json /home/model-server/
|
||||
COPY ./model/finetuned-bert-classifier/ /home/model-server/
|
||||
|
||||
@@ -21,7 +24,7 @@ EXPOSE 7080
|
||||
EXPOSE 7081
|
||||
|
||||
# create model archive file packaging model artifacts and dependencies
|
||||
RUN torch-model-archiver -f --model-name=finetuned-bert-classifier --version=1.0 --serialized-file=/home/model-server/pytorch_model.bin --handler=/home/model-server/custom_text_handler.py --extra-files "/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json" --export-path=/home/model-server/model-store
|
||||
RUN torch-model-archiver -f --model-name=finetuned-bert-classifier --version=1.0 --serialized-file=/home/model-server/pytorch_model.bin --handler=/home/model-server/custom_handler.py --extra-files "/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json" --export-path=/home/model-server/model-store
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "finetuned-bert-classifier=finetuned-bert-classifier.mar", "--model-store", "/home/model-server/model-store"]
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
|
||||
FROM pytorch/torchserve:latest-cpu
|
||||
|
||||
USER root
|
||||
# run and update some basic packages software packages, including security libs
|
||||
RUN apt-get update && apt-get install -y software-properties-common && add-apt-repository -y ppa:ubuntu-toolchain-r/test && apt-get update && apt-get install -y gcc-9 g++-9 apt-transport-https ca-certificates gnupg curl
|
||||
|
||||
# Install gcloud tools for gsutil as well as debugging
|
||||
RUN echo "deb [signed-by=/usr/share/keyrings/cloud.google.gpg] http://packages.cloud.google.com/apt cloud-sdk main" | tee -a /etc/apt/sources.list.d/google-cloud-sdk.list && curl https://packages.cloud.google.com/apt/doc/apt-key.gpg | apt-key --keyring /usr/share/keyrings/cloud.google.gpg add - && apt-get update -y && apt-get install google-cloud-sdk -y
|
||||
|
||||
USER model-server
|
||||
|
||||
# install dependencies
|
||||
RUN python3 -m pip install --upgrade pip
|
||||
RUN pip3 install transformers
|
||||
|
||||
ARG MODEL_NAME=finetuned-bert-classifier
|
||||
ENV MODEL_NAME="${MODEL_NAME}"
|
||||
|
||||
# health and prediction listener ports
|
||||
ARG AIP_HTTP_PORT=7080
|
||||
ENV AIP_HTTP_PORT="${AIP_HTTP_PORT}"
|
||||
|
||||
ARG MODEL_MGMT_PORT=7081
|
||||
|
||||
# expose health and prediction listener ports from the image
|
||||
EXPOSE "${AIP_HTTP_PORT}"
|
||||
EXPOSE "${MODEL_MGMT_PORT}"
|
||||
EXPOSE 8080 8081 8082 7070 7071
|
||||
|
||||
# create torchserve configuration file
|
||||
USER root
|
||||
RUN echo "service_envelope=json\n" "inference_address=http://0.0.0.0:${AIP_HTTP_PORT}\n" "management_address=http://0.0.0.0:${MODEL_MGMT_PORT}" >> /home/model-server/config.properties
|
||||
USER model-server
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["echo", "AIP_STORAGE_URI=${AIP_STORAGE_URI}", ";", "gsutil", "cp", "-r", "${AIP_STORAGE_URI}/${MODEL_NAME}.mar", "/home/model-server/model-store/", ";", "ls", "-ltr", "/home/model-server/model-store/", ";", "torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "${MODEL_NAME}=${MODEL_NAME}.mar", "--model-store", "/home/model-server/model-store"]
|
||||
@@ -52,7 +52,8 @@ class TransformersClassifierHandler(BaseHandler):
|
||||
with open(mapping_file_path) as f:
|
||||
self.mapping = json.load(f)
|
||||
else:
|
||||
logger.warning('Missing the index_to_name.json file. Inference output will not include class name.')
|
||||
logger.warning('Missing the index_to_name.json file. Inference output will default.')
|
||||
self.mapping = {"0": "Negative", "1": "Positive"}
|
||||
|
||||
self.initialized = True
|
||||
|
||||
@@ -88,4 +89,3 @@ class TransformersClassifierHandler(BaseHandler):
|
||||
|
||||
def postprocess(self, inference_output):
|
||||
return inference_output
|
||||
|
||||
@@ -19,13 +19,19 @@ echo "Submitting Custom Job to Vertex AI to train PyTorch model"
|
||||
|
||||
# BUCKET_NAME: Change to your bucket name
|
||||
BUCKET_NAME="[your-bucket-name]" # <-- CHANGE TO YOUR BUCKET NAME
|
||||
BUCKET_NAME="cloud-ai-platform-2f444b6a-a742-444b-b91a-c7519f51bd77"
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The PyTorch image provided by Vertex AI Training.
|
||||
IMAGE_URI="us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-7:latest"
|
||||
|
||||
# JOB_NAME: the name of your job running on Vertex AI.
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-pkg-ar-"
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-pkg-ar"
|
||||
JOB_NAME=${JOB_PREFIX}-$(date +%Y%m%d%H%M%S)-custom-job
|
||||
|
||||
# REGION: select a region from https://cloud.google.com/vertex-ai/docs/general/locations#available_regions
|
||||
@@ -35,19 +41,21 @@ REGION="us-central1"
|
||||
# JOB_DIR: Where to store prepared package and upload output model.
|
||||
JOB_DIR=gs://${BUCKET_NAME}/${JOB_PREFIX}/model/${JOB_NAME}
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
# worker pool spec
|
||||
worker_pool_spec="\
|
||||
replica-count=1,\
|
||||
machine-type=n1-standard-8,\
|
||||
accelerator-type=NVIDIA_TESLA_V100,\
|
||||
accelerator-count=1,\
|
||||
executor-image-uri=${IMAGE_URI},\
|
||||
python-module=trainer.task,\
|
||||
local-package-path=../python_package/"
|
||||
|
||||
# Submit Custom Job to Vertex AI
|
||||
gcloud beta ai custom-jobs create \
|
||||
--display-name=${JOB_NAME} \
|
||||
--region ${REGION} \
|
||||
--python-package-uris=${PACKAGE_PATH} \
|
||||
--worker-pool-spec=replica-count=1,machine-type='n1-standard-8',accelerator-type='NVIDIA_TESLA_V100',accelerator-count=1,executor-image-uri=${IMAGE_URI},python-module='trainer.task',local-package-path="../python_package/" \
|
||||
--worker-pool-spec="${worker_pool_spec}" \
|
||||
--args="--model-name","finetuned-bert-classifier","--job-dir",$JOB_DIR
|
||||
|
||||
echo "After the job is completed successfully, model files will be saved at $JOB_DIR/"
|
||||
|
||||
@@ -122,6 +122,9 @@ def run(args):
|
||||
# Train / Test the model
|
||||
trainer = train(args, text_classifier, train_dataset, test_dataset)
|
||||
|
||||
metrics = trainer.evaluate(eval_dataset=test_dataset)
|
||||
trainer.save_metrics("all", metrics)
|
||||
|
||||
# Export the trained model
|
||||
trainer.save_model(os.path.join("/tmp", args.model_name))
|
||||
|
||||
|
||||
@@ -63,20 +63,20 @@
|
||||
"- [Training](#Training)\n",
|
||||
" - [Run Training Locally in the Notebook](#Training-locally-in-the-notebook)\n",
|
||||
" - [Run Training Job on Vertex AI](#Training-on-Vertex-AI)\n",
|
||||
" - [Training with pre-built container](#Run-Custom-Job-on-Vertex-Training-with-a-pre-built-container)\n",
|
||||
" - [Training with custom container](#Run-Custom-Job-on-Vertex-Training-with-custom-container)\n",
|
||||
" - [Training with pre-built container](#Run-Custom-Job-on-Vertex-AI-Training-with-a-pre-built-container)\n",
|
||||
" - [Training with custom container](#Run-Custom-Job-on-Vertex-AI-Training-with-custom-container)\n",
|
||||
"- [Tuning](#Hyperparameter-Tuning) \n",
|
||||
" - [Run Hyperparameter Tuning job on Vertex AI](#Run-Hyperparameter-Tuning-Job-on-Vertex-AI)\n",
|
||||
"- [Deploying](#Deploying)\n",
|
||||
" - [Deploying model on Vertex Predictions with custom container](#Deploying-model-on-Vertex-Predictions-with-custom-container)\n",
|
||||
" - [Deploying model on Vertex AI Predictions with custom container](#Deploying-model-on-Vertex AI-Predictions-with-custom-container)\n",
|
||||
"\n",
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud Platform (GCP):\n",
|
||||
"\n",
|
||||
"* [Notebooks](https://cloud.google.com/notebooks)\n",
|
||||
"* [Vertex Training](https://cloud.google.com/vertex-ai/docs/training/custom-training)\n",
|
||||
"* [Vertex Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions)\n",
|
||||
"* [Vertex AI Workbench](https://cloud.google.com/vertex-ai-workbench)\n",
|
||||
"* [Vertex AI Training](https://cloud.google.com/vertex-ai/docs/training/custom-training)\n",
|
||||
"* [Vertex AI Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions)\n",
|
||||
"* [Cloud Storage](https://cloud.google.com/storage)\n",
|
||||
"* [Container Registry](https://cloud.google.com/container-registry)\n",
|
||||
"* [Cloud Build](https://cloud.google.com/build) *[Optional]*\n",
|
||||
@@ -202,9 +202,9 @@
|
||||
"id": "e0c1dcadc2c8"
|
||||
},
|
||||
"source": [
|
||||
"We will be using [Vertex SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) to interact with Vertex AI services. The high-level `aiplatform` library is designed to simplify common data science workflows by using wrapper classes and opinionated defaults. \n",
|
||||
"We will be using [Vertex AI SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) to interact with Vertex AI services. The high-level `aiplatform` library is designed to simplify common data science workflows by using wrapper classes and opinionated defaults. \n",
|
||||
"\n",
|
||||
"#### Install Vertex SDK for Python"
|
||||
"#### Install Vertex AI SDK for Python"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1199,7 +1199,7 @@
|
||||
"source": [
|
||||
"### Run predictions locally with sample examples\n",
|
||||
"\n",
|
||||
"Using the trained model, we can predict the sentiment label for an input text after applying the preprocessing function that was used during the training. We will run the predictions locally in the notebook and later show how you can deploy the model to an endpoint using [TorchServe](https://pytorch.org/serve/) on Vertex Predictions."
|
||||
"Using the trained model, we can predict the sentiment label for an input text after applying the preprocessing function that was used during the training. We will run the predictions locally in the notebook and later show how you can deploy the model to an endpoint using [TorchServe](https://pytorch.org/serve/) on Vertex AI Predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1382,7 +1382,7 @@
|
||||
"id": "f7466d414a0e"
|
||||
},
|
||||
"source": [
|
||||
"### Run Custom Job on Vertex Training with a pre-built container"
|
||||
"### Run Custom Job on Vertex AI Training with a pre-built container"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1395,7 +1395,7 @@
|
||||
"\n",
|
||||
"In this notebook, we are using Hugging Face Datasets and fine tuning a transformer model from Hugging Face Transformers Library for sentiment analysis task using PyTorch. We will use [pre-built container for PyTorch](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers#pytorch) and package the training application code by adding standard Python dependencies - `transformers`, `datasets` and `tqdm` - in the `setup.py` file. \n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1569,7 +1569,7 @@
|
||||
"source": [
|
||||
"#### **Run custom training job on Vertex AI**\n",
|
||||
"\n",
|
||||
"We use [Vertex SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#client_libraries) to create and submit training job to the Vertex training service."
|
||||
"We use [Vertex AI SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#client_libraries) to create and submit training job to the Vertex AI training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1578,7 +1578,7 @@
|
||||
"id": "5d2957ef04fd"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1598,7 +1598,7 @@
|
||||
"id": "6b0fed34b728"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Custom Job to Vertex Training service**"
|
||||
"##### **Configure and submit Custom Job to Vertex AI Training service**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1609,7 +1609,7 @@
|
||||
"source": [
|
||||
"Configure a [Custom Job](https://cloud.google.com/vertex-ai/docs/training/create-custom-job) with the [pre-built container](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers) image for PyTorch and training code packaged as Python source distribution. \n",
|
||||
"\n",
|
||||
"**NOTE:** When using Vertex SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job on Vertex Training service."
|
||||
"**NOTE:** When using Vertex AI SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job on Vertex AI Training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1686,7 +1686,7 @@
|
||||
"\n",
|
||||
"You can monitor the custom job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/training-pipelines/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1798,7 +1798,7 @@
|
||||
"id": "c170d386492b"
|
||||
},
|
||||
"source": [
|
||||
"### Run Custom Job on Vertex Training with custom container"
|
||||
"### Run Custom Job on Vertex AI Training with custom container"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1807,7 +1807,7 @@
|
||||
"id": "035227b6e581"
|
||||
},
|
||||
"source": [
|
||||
"To create a [training job with custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container?hl=hr), you define a `Dockerfile` to install or add the dependencies required for the training job. Then, you build and test your Docker image locally to verify, push the image to Container Registry and submit a Custom Job to Vertex Training service.\n",
|
||||
"To create a [training job with custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container?hl=hr), you define a `Dockerfile` to install or add the dependencies required for the training job. Then, you build and test your Docker image locally to verify, push the image to Container Registry and submit a Custom Job to Vertex AI Training service.\n",
|
||||
"\n",
|
||||
""
|
||||
]
|
||||
@@ -1834,7 +1834,7 @@
|
||||
"%%writefile ./custom_container/Dockerfile\n",
|
||||
"\n",
|
||||
"# Use pytorch GPU base image\n",
|
||||
"FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7\n",
|
||||
"FROM us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-10:latest\n",
|
||||
"\n",
|
||||
"# set working directory\n",
|
||||
"WORKDIR /app\n",
|
||||
@@ -1968,7 +1968,7 @@
|
||||
"id": "a23e5e34bea9"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1988,11 +1988,11 @@
|
||||
"id": "abf1fa4085cb"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Custom Job to Vertex Training service**\n",
|
||||
"##### **Configure and submit Custom Job to Vertex AI Training service**\n",
|
||||
"\n",
|
||||
"Configure a [Custom Job](https://cloud.google.com/vertex-ai/docs/training/create-custom-job) with the [custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container) image with training code and other dependencies\n",
|
||||
"\n",
|
||||
"**NOTE:** When using Vertex SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job to train on Vertex Training."
|
||||
"**NOTE:** When using Vertex AI SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job to train on Vertex AI Training."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2044,7 +2044,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# submit the custom job to Vertex training service\n",
|
||||
"# submit the custom job to Vertex AI training service\n",
|
||||
"model = job.run(\n",
|
||||
" replica_count=1,\n",
|
||||
" machine_type=\"n1-standard-8\",\n",
|
||||
@@ -2065,7 +2065,7 @@
|
||||
"\n",
|
||||
"You can monitor the custom job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/training-pipelines/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2148,11 +2148,11 @@
|
||||
"id": "ba6122f929e3"
|
||||
},
|
||||
"source": [
|
||||
"The training application code for fine-tuning a transformer model for sentiment analysis task uses hyperparameters such as learning rate and weight decay. These hyperparameters control the behavior of the training algorithm and can have a significant effect on the performance of the resulting model. This part of the notebook show how you can automate tuning these hyperparameters with Vertex Training service.\n",
|
||||
"The training application code for fine-tuning a transformer model for sentiment analysis task uses hyperparameters such as learning rate and weight decay. These hyperparameters control the behavior of the training algorithm and can have a significant effect on the performance of the resulting model. This part of the notebook show how you can automate tuning these hyperparameters with Vertex AI Training service.\n",
|
||||
"\n",
|
||||
"We submit a [Hyperparameter Tuning job](https://cloud.google.com/vertex-ai/docs/training/hyperparameter-tuning-overview) to Vertex Training service by packaging the training application code and dependencies in a Docker container and push the container to Google Container Registry, similar to running a Custom Job on Vertex AI with Custom Container.\n",
|
||||
"We submit a [Hyperparameter Tuning job](https://cloud.google.com/vertex-ai/docs/training/hyperparameter-tuning-overview) to Vertex AI Training service by packaging the training application code and dependencies in a Docker container and push the container to Google Container Registry, similar to running a Custom Job on Vertex AI with Custom Container.\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2163,7 +2163,7 @@
|
||||
"source": [
|
||||
"### How hyperparameter tuning works in Vertex AI?\n",
|
||||
"\n",
|
||||
"Following are the high level steps involved in running a Hyperparameter Tuning job on Vertex Training service:\n",
|
||||
"Following are the high level steps involved in running a Hyperparameter Tuning job on Vertex AI Training service:\n",
|
||||
"\n",
|
||||
"- You define the hyperparameters to tune the model along with the metric (or goal) to optimize\n",
|
||||
"- Vertex AI runs multiple trials of your training application with the hyperparameters and limits you specified - maximum number of trials to run and number of parallel trials. \n",
|
||||
@@ -2297,7 +2297,7 @@
|
||||
"source": [
|
||||
"### Run Hyperparameter Tuning Job on Vertex AI\n",
|
||||
"\n",
|
||||
"Before submitting the hyperparameter tuning job to Vertex AI, push the custom container image with training application to Google Cloud Container Registry and then submit the job to Vertex AI. We will be using the same image used for running Custom Job on Vertex Training service."
|
||||
"Before submitting the hyperparameter tuning job to Vertex AI, push the custom container image with training application to Google Cloud Container Registry and then submit the job to Vertex AI. We will be using the same image used for running Custom Job on Vertex AI Training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2326,7 +2326,7 @@
|
||||
"id": "f60fab07d67c"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2346,7 +2346,7 @@
|
||||
"id": "6652aa63ddff"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Hyperparameter Tuning Job to Vertex Training service**\n",
|
||||
"##### **Configure and submit Hyperparameter Tuning Job to Vertex AI Training service**\n",
|
||||
"\n",
|
||||
"Configure a [Hyperparameter Tuning Job](https://cloud.google.com/vertex-ai/docs/training/using-hyperparameter-tuning) with the [custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container) image with training code and other dependencies.\n",
|
||||
"\n",
|
||||
@@ -2374,7 +2374,7 @@
|
||||
"id": "9d46db3a8b23"
|
||||
},
|
||||
"source": [
|
||||
"Define the training arguments with `hp-tune` argument set to `y` so that training application code can report metrics to Vertex"
|
||||
"Define the training arguments with `hp-tune` argument set to `y` so that training application code can report metrics to Vertex AI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2548,7 +2548,7 @@
|
||||
"\n",
|
||||
"You can monitor the hyperparameter tuning job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/hyperparameter-tuning-jobs/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2557,7 +2557,7 @@
|
||||
"id": "ba934b434f03"
|
||||
},
|
||||
"source": [
|
||||
"After the job is finished, you can view and format the results of the hyperparameter tuning Trials (run by Vertex Training service) as a Pandas dataframe"
|
||||
"After the job is finished, you can view and format the results of the hyperparameter tuning Trials (run by Vertex AI Training service) as a Pandas dataframe"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2612,7 +2612,7 @@
|
||||
"id": "5dbccb2b7d32"
|
||||
},
|
||||
"source": [
|
||||
"Now from the results of Trials, you can pick the best performing Trial to deploy to Vertex Predictions"
|
||||
"Now from the results of Trials, you can pick the best performing Trial to deploy to Vertex AI Predictions"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2701,8 +2701,8 @@
|
||||
"JOB_NAME=${JOB_PREFIX}-pytorch-hptune-$(date +%Y%m%d%H%M%S)\n",
|
||||
"echo \"Launching hyperparameter tuning job with display name as \"$JOB_NAME\n",
|
||||
"\n",
|
||||
"# BUCKET_NAME: Change to your bucket name\n",
|
||||
"BUCKET_NAME=$1 # <-- CHANGE TO YOUR BUCKET NAME\n",
|
||||
"# BUCKET_NAME is a required parameter to run the cell.\n",
|
||||
"BUCKET_NAME=$1\n",
|
||||
"\n",
|
||||
"# APP_NAME: get application name\n",
|
||||
"APP_NAME=$2\n",
|
||||
@@ -2711,7 +2711,7 @@
|
||||
"JOB_DIR=${BUCKET_NAME}/${JOB_PREFIX}/model/${JOB_NAME}\n",
|
||||
"\n",
|
||||
"# custom container image URI\n",
|
||||
"CUSTOM_TRAIN_IMAGE_URI=f'gcr.io/'${PROJECT_ID}'/pytorch_gpu_train_'${APP_NAME}\n",
|
||||
"CUSTOM_TRAIN_IMAGE_URI='gcr.io/'${PROJECT_ID}'/pytorch_gpu_train_'${APP_NAME}\n",
|
||||
"\n",
|
||||
"# ========================================================\n",
|
||||
"# create hyperparameter tuning configuration file\n",
|
||||
@@ -2772,20 +2772,20 @@
|
||||
"source": [
|
||||
"## Deploying\n",
|
||||
"\n",
|
||||
"Deploying a PyTorch model on [Vertex Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions) requires to use a custom container that serves online predictions. You will deploy a container running [PyTorch's TorchServe](https://pytorch.org/serve/) tool in order to serve predictions from a fine-tuned transformer model from Hugging Face Transformers for sentiment analysis task. You can then use Vertex Predictions to classify sentiment of input texts. \n",
|
||||
"Deploying a PyTorch model on [Vertex AI Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions) requires to use a custom container that serves online predictions. You will deploy a container running [PyTorch's TorchServe](https://pytorch.org/serve/) tool in order to serve predictions from a fine-tuned transformer model from Hugging Face Transformers for sentiment analysis task. You can then use Vertex AI Predictions to classify sentiment of input texts. \n",
|
||||
"\n",
|
||||
"### Deploying model on Vertex Predictions with custom container\n",
|
||||
"### Deploying model on Vertex AI Predictions with custom container\n",
|
||||
"\n",
|
||||
"To use a custom container to serve predictions from a PyTorch model, you must provide Vertex AI with a Docker container image that runs an HTTP server, such as TorchServe in this case. Please refer to [documentation](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) that describes the container image requirements to be compatible with Vertex Predictions.\n",
|
||||
"To use a custom container to serve predictions from a PyTorch model, you must provide Vertex AI with a Docker container image that runs an HTTP server, such as TorchServe in this case. Please refer to [documentation](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) that describes the container image requirements to be compatible with Vertex AI Predictions.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Essentially, to deploy a PyTorch model on Vertex Predictions following are the steps:\n",
|
||||
"Essentially, to deploy a PyTorch model on Vertex AI Predictions following are the steps:\n",
|
||||
"\n",
|
||||
"1. Package the trained model artifacts including [default](https://pytorch.org/serve/#default-handlers) or [custom](https://pytorch.org/serve/custom_service.html) handlers by creating an archive file using [Torch model archiver](https://github.com/pytorch/serve/tree/master/model-archiver)\n",
|
||||
"2. Build a [custom container](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) compatible with Vertex Predictions to serve the model using Torchserve\n",
|
||||
"3. Upload the model with custom container image to serve predictions as a Vertex Model resource\n",
|
||||
"4. Create a Vertex Endpoint and [deploy the model](https://cloud.google.com/vertex-ai/docs/predictions/deploy-model-api) resource"
|
||||
"2. Build a [custom container](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) compatible with Vertex AI Predictions to serve the model using Torchserve\n",
|
||||
"3. Upload the model with custom container image to serve predictions as a Vertex AI Model resource\n",
|
||||
"4. Create a Vertex AI Endpoint and [deploy the model](https://cloud.google.com/vertex-ai/docs/predictions/deploy-model-api) resource"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2815,7 +2815,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile predictor/custom_text_handler.py\n",
|
||||
"%%writefile predictor/custom_handler.py\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import json\n",
|
||||
@@ -2870,7 +2870,8 @@
|
||||
" with open(mapping_file_path) as f:\n",
|
||||
" self.mapping = json.load(f)\n",
|
||||
" else:\n",
|
||||
" logger.warning('Missing the index_to_name.json file. Inference output will not include class name.')\n",
|
||||
" logger.warning('Missing the index_to_name.json file. Inference output will default.')\n",
|
||||
" self.mapping = {\"0\": \"Negative\", \"1\": \"Positive\"}\n",
|
||||
"\n",
|
||||
" self.initialized = True\n",
|
||||
"\n",
|
||||
@@ -3047,10 +3048,13 @@
|
||||
"FROM pytorch/torchserve:latest-cpu\n",
|
||||
"\n",
|
||||
"# install dependencies\n",
|
||||
"RUN python3 -m pip install --upgrade pip\n",
|
||||
"RUN pip3 install transformers\n",
|
||||
"\n",
|
||||
"USER model-server\n",
|
||||
"\n",
|
||||
"# copy model artifacts, custom handler and other dependencies\n",
|
||||
"COPY ./custom_text_handler.py /home/model-server/\n",
|
||||
"COPY ./custom_handler.py /home/model-server/\n",
|
||||
"COPY ./index_to_name.json /home/model-server/\n",
|
||||
"COPY ./model/$APP_NAME/ /home/model-server/\n",
|
||||
"\n",
|
||||
@@ -3070,7 +3074,7 @@
|
||||
" --model-name=$APP_NAME \\\n",
|
||||
" --version=1.0 \\\n",
|
||||
" --serialized-file=/home/model-server/pytorch_model.bin \\\n",
|
||||
" --handler=/home/model-server/custom_text_handler.py \\\n",
|
||||
" --handler=/home/model-server/custom_handler.py \\\n",
|
||||
" --extra-files \"/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json\" \\\n",
|
||||
" --export-path=/home/model-server/model-store\n",
|
||||
"\n",
|
||||
@@ -3129,7 +3133,7 @@
|
||||
"source": [
|
||||
"#### **Run the container locally** ***[Optional]***\n",
|
||||
"\n",
|
||||
"Before push the container image to Container Registry to use it with Vertex Predictions, you can run it as a container in your local environment to verify that the server works as expected"
|
||||
"Before push the container image to Container Registry to use it with Vertex AI Predictions, you can run it as a container in your local environment to verify that the server works as expected"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3267,9 +3271,9 @@
|
||||
"id": "69477b3a00c0"
|
||||
},
|
||||
"source": [
|
||||
"#### **Deploying the serving container to Vertex Predictions**\n",
|
||||
"#### **Deploying the serving container to Vertex AI Predictions**\n",
|
||||
"\n",
|
||||
"We create a model resource on Vertex AI and deploy the model to a Vertex Endpoints. You must deploy a model to an endpoint before using the model. The deployed model runs the custom container image to serve predictions. "
|
||||
"We create a model resource on Vertex AI and deploy the model to a Vertex AI Endpoints. You must deploy a model to an endpoint before using the model. The deployed model runs the custom container image to serve predictions. "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3300,7 +3304,7 @@
|
||||
"id": "a3da91e19af4"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3437,7 +3441,7 @@
|
||||
"id": "bc4673478269"
|
||||
},
|
||||
"source": [
|
||||
"#### **Invoking the Endpoint with deployed Model using Vertex SDK to make predictions**"
|
||||
"#### **Invoking the Endpoint with deployed Model using Vertex AI SDK to make predictions**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3487,7 +3491,7 @@
|
||||
"source": [
|
||||
"##### **Formatting input for online prediction**\n",
|
||||
"\n",
|
||||
"For online prediction requests, the prediction input instances must be formatted as JSON with base64 encoding as shown here:\n",
|
||||
"This notebook uses [Torchserve's KServe based inference API](https://pytorch.org/serve/inference_api.html#kserve-inference-api) which is also [Vertex AI Predictions compatible format](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements#prediction). For online prediction requests, format the prediction input instances as JSON with base64 encoding as shown here:\n",
|
||||
"\n",
|
||||
"```\n",
|
||||
"[\n",
|
||||
@@ -3560,9 +3564,9 @@
|
||||
},
|
||||
"source": [
|
||||
"##### ***[Optional]*** **Make prediction requests using gcloud CLI**\n",
|
||||
"You can also call the Vertex Endpoint to make predictions using [`gcloud beta ai endpoints predict`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/endpoints/predict). \n",
|
||||
"You can also call the Vertex AI Endpoint to make predictions using [`gcloud beta ai endpoints predict`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/endpoints/predict). \n",
|
||||
"\n",
|
||||
"The following cell shows how to make a prediction request to Vertex Endpoints using `gcloud` CLI: "
|
||||
"The following cell shows how to make a prediction request to Vertex AI Endpoints using `gcloud` CLI: "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3653,12 +3657,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_custom_job = True\n",
|
||||
"delete_hp_tuning_job = True\n",
|
||||
"delete_custom_job = False\n",
|
||||
"delete_hp_tuning_job = False\n",
|
||||
"delete_endpoint = True\n",
|
||||
"delete_model = True\n",
|
||||
"delete_bucket = True\n",
|
||||
"delete_image = True"
|
||||
"delete_model = False\n",
|
||||
"delete_bucket = False\n",
|
||||
"delete_image = False"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3686,7 +3690,7 @@
|
||||
"\n",
|
||||
"client_options = {\"api_endpoint\": API_ENDPOINT}\n",
|
||||
"\n",
|
||||
"# Initialize Vertex SDK\n",
|
||||
"# Initialize Vertex AI SDK\n",
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
]
|
||||
},
|
||||
@@ -3924,7 +3928,7 @@
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" print(f\"Deleting all contents from the bucket {BUCKET_NAME}\")\n",
|
||||
"\n",
|
||||
" shell_output=! gsutil du -as $BUCKET_NAME\n",
|
||||
" shell_output = ! gsutil du -as $BUCKET_NAME\n",
|
||||
" print(\n",
|
||||
" f\"Size of the bucket {BUCKET_NAME} before deleting = {shell_output[0].split()[0]} bytes\"\n",
|
||||
" )\n",
|
||||
@@ -3932,7 +3936,7 @@
|
||||
" # uncomment below line to delete contents of the bucket\n",
|
||||
" # ! gsutil rm -r $BUCKET_NAME\n",
|
||||
"\n",
|
||||
" shell_output=! gsutil du -as $BUCKET_NAME\n",
|
||||
" shell_output = ! gsutil du -as $BUCKET_NAME\n",
|
||||
" if float(shell_output[0].split()[0]) > 0:\n",
|
||||
" print(\n",
|
||||
" \"PLEASE UNCOMMENT LINE TO DELETE BUCKET. CONTENT FROM THE BUCKET NOT DELETED\"\n",
|
||||
|
||||
@@ -188,11 +188,14 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install {USER_FLAG} google-cloud-aiplatform==1.0.1\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-pipeline-components==0.1.3\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-aiplatform\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-pipeline-components\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade kfp\n",
|
||||
"! pip3 install {USER_FLAG} numpy==1.20.3\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tensorflow"
|
||||
"! pip3 install {USER_FLAG} numpy\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tensorflow\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade pillow\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tf-agents\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade fastapi"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -287,7 +290,7 @@
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
@@ -518,6 +521,7 @@
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"from kfp.v2 import compiler, dsl\n",
|
||||
"from kfp.v2.google.client import AIPlatformClient"
|
||||
@@ -561,13 +565,34 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H3530hdGGilo"
|
||||
"id": "895ac243c125"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Dataset parameters\n",
|
||||
"RAW_DATA_PATH = \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" # Location of the MovieLens 100K dataset's \"u.data\" file.\n",
|
||||
"\n",
|
||||
"RAW_DATA_PATH = \"gs://[your-bucket-name]/raw_data/u.data\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "62bfb9a820f6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download the sample data into your RAW_DATA_PATH\n",
|
||||
"! gsutil cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $RAW_DATA_PATH"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H3530hdGGilo"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Pipeline parameters\n",
|
||||
"PIPELINE_NAME = \"movielens-pipeline\" # Pipeline display name.\n",
|
||||
"ENABLE_CACHING = False # Whether to enable execution caching for the pipeline.\n",
|
||||
@@ -635,7 +660,7 @@
|
||||
"source": [
|
||||
"#### Run unit tests on the Generator component\n",
|
||||
"\n",
|
||||
"Before running the command, fill in `RAW_DATA_PATH` in [`src/generator/test_generator_component.py`](src/generator/test_generator_component.py)."
|
||||
"Before running the command, you should update the `RAW_DATA_PATH` in [`src/generator/test_generator_component.py`](src/generator/test_generator_component.py)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -713,12 +738,12 @@
|
||||
"TRAINING_ARTIFACTS_DIR = (\n",
|
||||
" f\"{BUCKET_NAME}/artifacts\" # Root directory for training artifacts.\n",
|
||||
")\n",
|
||||
"TRAINING_REPLICA_COUNT = \"1\" # Number of replica to run the custom training job.\n",
|
||||
"TRAINING_REPLICA_COUNT = 1 # Number of replica to run the custom training job.\n",
|
||||
"TRAINING_MACHINE_TYPE = (\n",
|
||||
" \"n1-standard-4\" # Type of machine to run the custom training job.\n",
|
||||
")\n",
|
||||
"TRAINING_ACCELERATOR_TYPE = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Type of accelerators to run the custom training job.\n",
|
||||
"TRAINING_ACCELERATOR_COUNT = \"0\" # Number of accelerators for the custom training job."
|
||||
"TRAINING_ACCELERATOR_COUNT = 0 # Number of accelerators for the custom training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -769,8 +794,12 @@
|
||||
"TRAINED_POLICY_DISPLAY_NAME = (\n",
|
||||
" \"movielens-trained-policy\" # Display name of the uploaded and deployed policy.\n",
|
||||
")\n",
|
||||
"TRAFFIC_SPLIT = {\"0\": 100}\n",
|
||||
"ENDPOINT_DISPLAY_NAME = \"movielens-endpoint\" # Display name of the prediction endpoint.\n",
|
||||
"ENDPOINT_MACHINE_TYPE = \"n1-standard-4\" # Type of machine of the prediction endpoint."
|
||||
"ENDPOINT_MACHINE_TYPE = \"n1-standard-4\" # Type of machine of the prediction endpoint.\n",
|
||||
"ENDPOINT_REPLICA_COUNT = 1 # Number of replicas of the prediction endpoint.\n",
|
||||
"ENDPOINT_ACCELERATOR_TYPE = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Type of accelerators to run the custom training job.\n",
|
||||
"ENDPOINT_ACCELERATOR_COUNT = 0 # Number of accelerators for the custom training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -900,16 +929,17 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components.experimental.custom_job import utils\n",
|
||||
"from kfp.components import load_component_from_url\n",
|
||||
"\n",
|
||||
"generate_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/generator/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/generator/component.yaml\"\n",
|
||||
")\n",
|
||||
"ingest_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
")\n",
|
||||
"train_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -978,7 +1008,7 @@
|
||||
" bigquery_location=bigquery_location,\n",
|
||||
" bigquery_table_id=bigquery_table_id,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" \n",
|
||||
" # Run the Ingester component.\n",
|
||||
" ingest_task = ingest_op(\n",
|
||||
" project_id=project_id,\n",
|
||||
@@ -988,7 +1018,16 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Run the Trainer component and submit custom job to Vertex AI.\n",
|
||||
" train_task = train_op(\n",
|
||||
" # Convert the train_op component into a Vertex AI Custom Job pre-built component\n",
|
||||
" custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" component_spec=train_op,\n",
|
||||
" replica_count=TRAINING_REPLICA_COUNT,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train_task = custom_job_training_op(\n",
|
||||
" training_artifacts_dir=training_artifacts_dir,\n",
|
||||
" tfrecord_file=ingest_task.outputs[\"tfrecord_file\"],\n",
|
||||
" num_epochs=num_epochs,\n",
|
||||
@@ -996,28 +1035,10 @@
|
||||
" num_actions=num_actions,\n",
|
||||
" tikhonov_weight=tikhonov_weight,\n",
|
||||
" agent_alpha=agent_alpha,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" worker_pool_specs = [\n",
|
||||
" {\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": train_task.container.image,\n",
|
||||
" },\n",
|
||||
" \"replicaCount\": TRAINING_REPLICA_COUNT,\n",
|
||||
" \"machineSpec\": {\n",
|
||||
" \"machineType\": TRAINING_MACHINE_TYPE,\n",
|
||||
" \"acceleratorType\": TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" \"acceleratorCount\": TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" train_task.custom_job_spec = {\n",
|
||||
" \"displayName\": train_task.name,\n",
|
||||
" \"jobSpec\": {\n",
|
||||
" \"workerPoolSpecs\": worker_pool_specs,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Run the Deployer components.\n",
|
||||
" # Upload the trained policy as a model.\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
@@ -1034,11 +1055,14 @@
|
||||
" # Deploy the uploaded, trained policy to the created endpoint. (This operation\n",
|
||||
" # has to occur after both model uploading and endpoint creation complete.)\n",
|
||||
" gcc_aip.ModelDeployOp(\n",
|
||||
" project=project_id,\n",
|
||||
" endpoint=endpoint_create_op.outputs[\"endpoint\"],\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" deployed_model_display_name=TRAINED_POLICY_DISPLAY_NAME,\n",
|
||||
" machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" traffic_split=TRAFFIC_SPLIT,\n",
|
||||
" dedicated_resources_machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_accelerator_type=ENDPOINT_ACCELERATOR_TYPE,\n",
|
||||
" dedicated_resources_accelerator_count=ENDPOINT_ACCELERATOR_COUNT,\n",
|
||||
" dedicated_resources_min_replica_count=ENDPOINT_REPLICA_COUNT,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
@@ -1053,12 +1077,11 @@
|
||||
"# Compile the authored pipeline.\n",
|
||||
"compiler.Compiler().compile(pipeline_func=pipeline, package_path=PIPELINE_SPEC_PATH)\n",
|
||||
"\n",
|
||||
"# Createa Vertex AI client.\n",
|
||||
"api_client = AIPlatformClient(project_id=PROJECT_ID, region=REGION)\n",
|
||||
"\n",
|
||||
"# Create a pipeline run job.\n",
|
||||
"response = api_client.create_run_from_job_spec(\n",
|
||||
" job_spec_path=PIPELINE_SPEC_PATH,\n",
|
||||
"job = aiplatform.PipelineJob(\n",
|
||||
" display_name=f\"{PIPELINE_NAME}-startup\",\n",
|
||||
" template_path=PIPELINE_SPEC_PATH,\n",
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" # Pipeline configs\n",
|
||||
" \"project_id\": PROJECT_ID,\n",
|
||||
@@ -1070,7 +1093,9 @@
|
||||
" \"bigquery_table_id\": BIGQUERY_TABLE_ID,\n",
|
||||
" },\n",
|
||||
" enable_caching=ENABLE_CACHING,\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"job.run()"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1111,7 +1136,11 @@
|
||||
"SIMULATOR_SCHEDULE = \"*/5 * * * *\" # Cloud Scheduler cron job schedule for the Simulator. Eg. \"*/5 * * * *\" means every 5 mins.\n",
|
||||
"SIMULATOR_SCHEDULER_MESSAGE = (\n",
|
||||
" \"simulator-message\" # Cloud Scheduler message for the Simulator.\n",
|
||||
")"
|
||||
")\n",
|
||||
"# TF-Agents RL configs\n",
|
||||
"BATCH_SIZE = 8\n",
|
||||
"RANK_K = 20\n",
|
||||
"NUM_ACTIONS = 20"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1221,7 +1250,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoints = ! gcloud beta ai endpoints list \\\n",
|
||||
"endpoints = ! gcloud ai endpoints list \\\n",
|
||||
" --region=$REGION \\\n",
|
||||
" --filter=display_name=$ENDPOINT_DISPLAY_NAME\n",
|
||||
"print(\"\\n\".join(endpoints), \"\\n\")\n",
|
||||
@@ -1424,13 +1453,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from kfp.components import load_component_from_url\n",
|
||||
"\n",
|
||||
"ingest_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
")\n",
|
||||
"train_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -1481,7 +1508,16 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Run the Trainer component and submit custom job to Vertex AI.\n",
|
||||
" train_task = train_op(\n",
|
||||
" # Convert the train_op component into a Vertex AI Custom Job pre-built component\n",
|
||||
" custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" component_spec=train_op,\n",
|
||||
" replica_count=TRAINING_REPLICA_COUNT,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train_task = custom_job_training_op(\n",
|
||||
" training_artifacts_dir=training_artifacts_dir,\n",
|
||||
" tfrecord_file=ingest_task.outputs[\"tfrecord_file\"],\n",
|
||||
" num_epochs=num_epochs,\n",
|
||||
@@ -1489,28 +1525,10 @@
|
||||
" num_actions=num_actions,\n",
|
||||
" tikhonov_weight=tikhonov_weight,\n",
|
||||
" agent_alpha=agent_alpha,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" worker_pool_specs = [\n",
|
||||
" {\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": train_task.container.image,\n",
|
||||
" },\n",
|
||||
" \"replicaCount\": TRAINING_REPLICA_COUNT,\n",
|
||||
" \"machineSpec\": {\n",
|
||||
" \"machineType\": TRAINING_MACHINE_TYPE,\n",
|
||||
" \"acceleratorType\": TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" \"acceleratorCount\": TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" train_task.custom_job_spec = {\n",
|
||||
" \"displayName\": train_task.name,\n",
|
||||
" \"jobSpec\": {\n",
|
||||
" \"workerPoolSpecs\": worker_pool_specs,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Run the Deployer components.\n",
|
||||
" # Upload the trained policy as a model.\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
@@ -1527,11 +1545,13 @@
|
||||
" # Deploy the uploaded, trained policy to the created endpoint. (This operation\n",
|
||||
" # has to occur after both model uploading and endpoint creation complete.)\n",
|
||||
" gcc_aip.ModelDeployOp(\n",
|
||||
" project=project_id,\n",
|
||||
" endpoint=endpoint_create_op.outputs[\"endpoint\"],\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" deployed_model_display_name=TRAINED_POLICY_DISPLAY_NAME,\n",
|
||||
" machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_accelerator_type=ENDPOINT_ACCELERATOR_TYPE,\n",
|
||||
" dedicated_resources_accelerator_count=ENDPOINT_ACCELERATOR_COUNT,\n",
|
||||
" dedicated_resources_min_replica_count=ENDPOINT_REPLICA_COUNT,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -39,14 +39,15 @@ outputs:
|
||||
- {name: bigquery_table_id, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- (PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'google-cloud-bigquery==2.20.0' 'tensorflow==2.5.0' 'tf-agents==0.8.0' || PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||
python3 -m pip install --quiet --no-warn-script-location 'google-cloud-bigquery==2.20.0'
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' --user) && "$0" "$@"
|
||||
'google-cloud-bigquery==2.20.0' 'pillow' 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
|| PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'google-cloud-bigquery==2.20.0' 'pillow' 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
--user) && "$0" "$@"
|
||||
- sh
|
||||
- -ec
|
||||
- |
|
||||
@@ -296,7 +297,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -20,7 +20,7 @@ outputs:
|
||||
- {name: tfrecord_file, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
@@ -187,7 +187,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
google-cloud-bigquery==2.20.0
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.7.2
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
|
||||
@@ -1,2 +1,4 @@
|
||||
google-cloud-pubsub==2.5.0
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.7.2
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
dataclasses==0.6
|
||||
google-cloud-aiplatform==1.8.1
|
||||
tensorflow==2.7.2
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
@@ -27,14 +27,14 @@ outputs:
|
||||
- {name: training_artifacts_dir, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- (PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' || PIP_DISABLE_PIP_VERSION_CHECK=1 python3
|
||||
-m pip install --quiet --no-warn-script-location 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
--user) && "$0" "$@"
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' 'Pillow' || PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||
python3 -m pip install --quiet --no-warn-script-location 'tensorflow==2.5.0'
|
||||
'tf-agents==0.8.0' 'Pillow' --user) && "$0" "$@"
|
||||
- sh
|
||||
- -ec
|
||||
- |
|
||||
@@ -270,7 +270,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -22,13 +22,13 @@ from src.training import task
|
||||
|
||||
|
||||
# Paths and configurations
|
||||
DATA_PATH = "gs://[your-bucket-name]/[your-dataset-dir]/u.data" # FILL IN
|
||||
DATA_PATH = "gs://[your-bucket-name]/artifacts/u.data" # FILL IN
|
||||
ROOT_DIR = "gs://[your-bucket-name]/artifacts" # FILL IN
|
||||
ARTIFACTS_DIR = "gs://[your-bucket-name]/artifacts" # FILL IN
|
||||
PROFILER_DIR = "gs://[your-bucket-name]/profiler" # FILL IN
|
||||
HPTUNING_RESULT_DIR = "[your-hptuning-result-dir]/" # FILL IN
|
||||
HPTUNING_RESULT_PATH = os.path.join(HPTUNING_RESULT_DIR,
|
||||
"[your-file-name].json") # FILL IN
|
||||
"result.json") # FILL IN
|
||||
RAW_BUCKET_NAME = "[your-hptuning-result-bucket-name]" # FILL IN
|
||||
|
||||
# Hyperparameters
|
||||
|
||||
@@ -1 +1 @@
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.7.2
|
||||
@@ -1 +1 @@
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.7.2
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -113,8 +113,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud beta ai custom-jobs local-run \\\n",
|
||||
" --base-image=$BASE_IMAGE_URI \\\n",
|
||||
"! gcloud ai custom-jobs local-run \\\n",
|
||||
" --executor-image-uri=$BASE_IMAGE_URI \\\n",
|
||||
" --script=$SCRIPT_PATH \\\n",
|
||||
" --output-image-uri=$OUTPUT_IMAGE_NAME \\\n",
|
||||
" -- \\\n",
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
The [official](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/official) folder contains notebooks organized by Google Cloud product. These are tested weekly and maintained by Google.
|
||||
|
||||
The [community](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/community) folder contains notebooks that may be created by Google or external contributors. They are not necessary maintained.
|
||||
|
||||
Contributions to the repo should use the [notebook template](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb) as a starting point.
|
||||
@@ -3,16 +3,24 @@
|
||||
# @global-owner1 and @global-owner2 will be requested for
|
||||
# review when someone opens a pull request.
|
||||
|
||||
/sdk/sdk_* @aferlitsch
|
||||
/gapic @aferlitsch
|
||||
/ml_ops @aferlitsch
|
||||
/model_monitoring/* @mco
|
||||
/sdk/sdk_* @andrewferlitsch
|
||||
/gapic @andrewferlitsch
|
||||
/ml_ops @andrewferlitsch
|
||||
/model_monitoring/* @mco-gh
|
||||
/structured_data/rapid_prototyping_* @rafael-carvalho
|
||||
|
||||
/managed_notebooks/ @notebooks-team
|
||||
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
|
||||
/managed_notebooks/
|
||||
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
|
||||
/pipelines/google_cloud_pipeline_components_TPU_model_train_upload_deploy.ipynb @brianchunkang
|
||||
/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb @thehardikv
|
||||
/sdk/sdk_automl_forecasting_evaluating_a_model.ipynb @thehardikv
|
||||
/matching_engine @yinghsienwu
|
||||
/neo4j @benofben @htappen
|
||||
/explainable_ai/SDK_Custom_Container_XAI.ipynb @brianchunkang
|
||||
/matching_engine/sdk_matching_engine_for_indexing.ipynb @ivanmkc
|
||||
/matching_engine/matching_engine_for_indexing.ipynb @yinghsienwu
|
||||
/sdk/pytorch_lightning_custom_container_training.ipynb @brianchunkang
|
||||
/tensorboard @yfang1
|
||||
/feature_store @nayaknishant @morgandu
|
||||
/vertex_endpoints/tf_hub_obj_detection/deploy_tfhub_object_detection_on_vertex_endpoints.ipynb @entrpn
|
||||
/vertex_endpoints/nvidia-triton/nvidia-triton-custom-container-prediction.ipynb @RajeshThallam
|
||||
/vertex_endpoints/optimized_tensorflow_runtime @vlasenkoalexey
|
||||
/notebooks/community/ml_ops/stage2/get_started_with_visionapi_and_automl.ipynb @mansari
|
||||
/notebooks/community/neo4j/graph_paysim.ipynb @benofben @laeg
|
||||
/notebooks/community/ml_ops/stage1/get_started_with_visionapi_and_vertex_datasets.ipynb @mansari
|
||||
|
||||
|
After Width: | Height: | Size: 83 KiB |
|
After Width: | Height: | Size: 141 KiB |
|
After Width: | Height: | Size: 230 KiB |
|
After Width: | Height: | Size: 140 KiB |
|
After Width: | Height: | Size: 140 KiB |
|
After Width: | Height: | Size: 382 KiB |
|
After Width: | Height: | Size: 445 KiB |
|
After Width: | Height: | Size: 63 KiB |
|
After Width: | Height: | Size: 59 KiB |
@@ -0,0 +1,823 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d1cc1c1fa076"
|
||||
},
|
||||
"source": [
|
||||
"# Pricing Optimization \n",
|
||||
"## Table of contents\n",
|
||||
"* [Overview](#section-1)\n",
|
||||
"* [Dataset](#section-2)\n",
|
||||
"* [Objective](#section-3)\n",
|
||||
"* [Costs](#section-4)\n",
|
||||
"* [Create a BigQuery dataset](#section-5)\n",
|
||||
"* [Load the dataset from Cloud Storage](#section-6)\n",
|
||||
"* [Data analysis](#section-7)\n",
|
||||
"* [Preprocess the data for training](#section-8)\n",
|
||||
"* [Train the model using BigQuery ML](#section-9)\n",
|
||||
"* [Generate forecasts from the model](#section-10)\n",
|
||||
"* [Interpret the results to choose the best price](#section-11)\n",
|
||||
"* [Clean up](#section-12)\n",
|
||||
"\n",
|
||||
"## Overview\n",
|
||||
"<a name=\"section-1\"></a>\n",
|
||||
"\n",
|
||||
"This notebook demonstrates analysis of pricing optimization on [CDM Pricing Data](https://github.com/trifacta/trifacta-google-cloud/tree/main/design-pattern-pricing-optimization) and automating the workflow using Vertex AI Workbench managed notebooks.\n",
|
||||
"\n",
|
||||
"*Note: This notebook file was developed to run in a [Vertex AI Workbench managed notebooks](https://console.cloud.google.com/vertex-ai/workbench/list/managed) instance using the Python (Local) kernel. Some components of this notebook may not work in other notebook environments.*\n",
|
||||
"\n",
|
||||
"## Dataset\n",
|
||||
"<a name=\"section-2\"></a>\n",
|
||||
"\n",
|
||||
"The dataset used in this notebook is a part of the [CDM Pricing dataset](https://github.com/trifacta/trifacta-google-cloud/blob/main/design-pattern-pricing-optimization/CDM_Pricing_large_table.csv), which consists of product sales information on specified dates.\n",
|
||||
"\n",
|
||||
"## Objective\n",
|
||||
"<a name=\"section-3\"></a>\n",
|
||||
"\n",
|
||||
"The objective of this notebook is to build a pricing optimization model using Vertex AI. The following steps have been followed: \n",
|
||||
"\n",
|
||||
"- Load the required dataset from a Cloud Storage bucket.\n",
|
||||
"- Analyze the fields present in the dataset.\n",
|
||||
"- Process the data to build a model.\n",
|
||||
"- Build a BigQuery ML forecast model on the processed data.\n",
|
||||
"- Get forecasted values from the BigQuery ML model.\n",
|
||||
"- Interpret the forecasts to identify the best prices.\n",
|
||||
"- Clean up.\n",
|
||||
"\n",
|
||||
"## Costs\n",
|
||||
"<a name=\"section-4\"></a>\n",
|
||||
"\n",
|
||||
"This tutorial uses the following billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- BigQuery\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing), [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5ed1f5e85640"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "c3f30148b66d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "750bf2883c2d"
|
||||
},
|
||||
"source": [
|
||||
"Otherwise, set your project ID here."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "3c6db1ca88b9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
|
||||
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2a1c270c7d34"
|
||||
},
|
||||
"source": [
|
||||
"### Import the required libraries and define constants\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "acc6fac1fa55"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import pandas as pd\n",
|
||||
"import seaborn as sns\n",
|
||||
"from google.cloud import bigquery\n",
|
||||
"from google.cloud.bigquery import Client"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a06006dff8f9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DATASET = \"[your-bigquery-dataset-id]\" # set the BigQuery dataset-id\n",
|
||||
"TRAINING_DATA_TABLE = \"[your-bigquery-table-id-to-store-the-training-data]\" # set the BigQuery table-id to store the training data"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "016c3d47cc69"
|
||||
},
|
||||
"source": [
|
||||
"## Create a BigQuery dataset\n",
|
||||
"<a name=\"section-5\"></a>\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "12ccd8d7956e"
|
||||
},
|
||||
"source": [
|
||||
"#@bigquery\n",
|
||||
"-- create a dataset in BigQuery\n",
|
||||
"\n",
|
||||
"CREATE SCHEMA pricing_optimization\n",
|
||||
"OPTIONS(\n",
|
||||
" location=\"us\"\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "c106b978a79b"
|
||||
},
|
||||
"source": [
|
||||
"## Load the dataset from Cloud Storage\n",
|
||||
"<a name=\"section-6\"></a>\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "8aeae9da9796"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DATA_LOCATION = \"gs://cloud-samples-data/ai-platform-unified/datasets/tabular/cdm_pricing_large_table.csv\"\n",
|
||||
"df = pd.read_csv(DATA_LOCATION)\n",
|
||||
"print(df.shape)\n",
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "7b98d5f09842"
|
||||
},
|
||||
"source": [
|
||||
"You will build a forecast model on this data and thus determine the best price for a product. For this type of model, you will not be using many fields: only the sales and price related ones. For the current execrcise, focus on the following fields:\n",
|
||||
"\n",
|
||||
"- `Product_ID`\n",
|
||||
"- `Customer_Hierarchy`\n",
|
||||
"- `Fiscal_Date`\n",
|
||||
"- `List_Price_Converged`\n",
|
||||
"- `Invoiced_quantity_in_Pieces`\n",
|
||||
"- `Net_Sales`\n",
|
||||
"\n",
|
||||
"## Data Analysis\n",
|
||||
"<a name=\"section-7\"></a>\n",
|
||||
"\n",
|
||||
"First, explore the data and distributions.\n",
|
||||
"\n",
|
||||
"Select the required columns from the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "af4b41c5eb1f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"id_col = \"Product_ID\"\n",
|
||||
"date_col = \"Fiscal_Date\"\n",
|
||||
"categ_cols = [\"Customer_Hierarchy\"]\n",
|
||||
"num_cols = [\"List_Price_Converged\", \"Invoiced_quantity_in_Pieces\", \"Net_Sales\"]\n",
|
||||
"\n",
|
||||
"df = df[[id_col, date_col] + categ_cols + num_cols].copy()\n",
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3d780043ee5b"
|
||||
},
|
||||
"source": [
|
||||
"Check the column types and null values in the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f54c445a1288"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df.info()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cd817b414c4d"
|
||||
},
|
||||
"source": [
|
||||
"This data description reveals that there are no null values in the data. Also, the field `Fiscal_Date` which is a date field is loaded as an object type. \n",
|
||||
"\n",
|
||||
"Change the type of the date field to datetime."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b160fac085c8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df[\"Fiscal_Date\"] = pd.to_datetime(df[\"Fiscal_Date\"], infer_datetime_format=True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fb4778578064"
|
||||
},
|
||||
"source": [
|
||||
"Plot the distributions for the categorical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "dd0467cd57c3"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in categ_cols:\n",
|
||||
" df[i].value_counts(normalize=True).plot(kind=\"bar\")\n",
|
||||
" plt.title(i)\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "145deed255e0"
|
||||
},
|
||||
"source": [
|
||||
"Plot the distributions for the numerical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f934137c6d82"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in num_cols:\n",
|
||||
" _, ax = plt.subplots(1, 2, figsize=(10, 4))\n",
|
||||
" df[i].plot(kind=\"box\", ax=ax[0])\n",
|
||||
" df[i].plot(kind=\"hist\", ax=ax[1])\n",
|
||||
" ax[0].set_title(i + \"-Boxplot\")\n",
|
||||
" ax[1].set_title(i + \"-Histogram\")\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f9b9c2e58380"
|
||||
},
|
||||
"source": [
|
||||
"Check the maximum date and minimum date in Fiscal_Date column."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2a10aa689f9d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"print(df[\"Fiscal_Date\"].max())\n",
|
||||
"print(df[\"Fiscal_Date\"].min())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "4834f63e2e59"
|
||||
},
|
||||
"source": [
|
||||
"Check the product distribution across each category."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4664877f5304"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"grp_cols = [\"Customer_Hierarchy\", \"Product_ID\"]\n",
|
||||
"grp_df = df[grp_cols].groupby(by=grp_cols).count().reset_index()\n",
|
||||
"grp_df.groupby(\"Customer_Hierarchy\").nunique()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "01ed02b9c8fd"
|
||||
},
|
||||
"source": [
|
||||
"Check the percentage changes in the orders based on the percentage changes in the price."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "0b2c428cb135"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# aggregate the data\n",
|
||||
"df_aggr = (\n",
|
||||
" df.groupby([\"Product_ID\", \"List_Price_Converged\"])\n",
|
||||
" .agg({\"Fiscal_Date\": min, \"Invoiced_quantity_in_Pieces\": sum, \"Net_Sales\": sum})\n",
|
||||
" .reset_index()\n",
|
||||
")\n",
|
||||
"# rename the aggregated columns\n",
|
||||
"df_aggr.rename(\n",
|
||||
" columns={\n",
|
||||
" \"Fiscal_Date\": \"First_price_date\",\n",
|
||||
" \"Invoiced_quantity_in_Pieces\": \"Total_ordered_pieces\",\n",
|
||||
" \"Net_Sales\": \"Total_net_sales\",\n",
|
||||
" },\n",
|
||||
" inplace=True,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# sort values chronologically\n",
|
||||
"df_aggr.sort_values(by=[\"Product_ID\", \"First_price_date\"], inplace=True)\n",
|
||||
"df_aggr.reset_index(drop=True, inplace=True)\n",
|
||||
"\n",
|
||||
"# add columns for previous values\n",
|
||||
"df_aggr[\"Previous_List\"] = df_aggr.groupby([\"Product_ID\"])[\n",
|
||||
" \"List_Price_Converged\"\n",
|
||||
"].shift()\n",
|
||||
"df_aggr[\"Previous_Total_ordered_pieces\"] = df_aggr.groupby([\"Product_ID\"])[\n",
|
||||
" \"Total_ordered_pieces\"\n",
|
||||
"].shift()\n",
|
||||
"\n",
|
||||
"# average price change across sku's\n",
|
||||
"df_aggr[\"price_change_perc\"] = (\n",
|
||||
" (df_aggr[\"List_Price_Converged\"] - df_aggr[\"Previous_List\"])\n",
|
||||
" / df_aggr[\"Previous_List\"].fillna(0)\n",
|
||||
" * 100\n",
|
||||
")\n",
|
||||
"df_aggr[\"order_change_perc\"] = (\n",
|
||||
" (df_aggr[\"Total_ordered_pieces\"] - df_aggr[\"Previous_Total_ordered_pieces\"])\n",
|
||||
" / df_aggr[\"Previous_Total_ordered_pieces\"].fillna(0)\n",
|
||||
" * 100\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# plot a scatterplot to visualize the changes\n",
|
||||
"sns.scatterplot(\n",
|
||||
" x=\"price_change_perc\",\n",
|
||||
" y=\"order_change_perc\",\n",
|
||||
" data=df_aggr,\n",
|
||||
" hue=\"Product_ID\",\n",
|
||||
" legend=False,\n",
|
||||
")\n",
|
||||
"plt.title(\"Percentage of change in price vs order\")\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "8259e916fe25"
|
||||
},
|
||||
"source": [
|
||||
"For most of the products, the percentage change in orders are high where the percentage changes in the prices are low. This suggests that too much change in the prices can affect the number of orders. \n",
|
||||
"\n",
|
||||
"**Note**: There seem to be some outliers in the data as percentage changes greater than 800 are found. In the current exercise, do not take any manual measures to deal with outliers as you will create a BigQuery ML timeseries model that already deals with outliers.\n",
|
||||
"\n",
|
||||
"## Preprocess the data for training\n",
|
||||
"<a name=\"section-8\"></a>\n",
|
||||
"\n",
|
||||
"Check which `Product_ID`'s have the maximum orders."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f5cbc7709c6a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_orders = df.groupby([\"Product_ID\", \"Customer_Hierarchy\"], as_index=False)[\n",
|
||||
" \"Invoiced_quantity_in_Pieces\"\n",
|
||||
"].sum()\n",
|
||||
"df_orders.loc[\n",
|
||||
" df_orders.groupby(\"Customer_Hierarchy\")[\"Invoiced_quantity_in_Pieces\"].idxmax()\n",
|
||||
"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fd6d227e513e"
|
||||
},
|
||||
"source": [
|
||||
"From the above result, you can infer the following:\n",
|
||||
"\n",
|
||||
"- Under the **Food** category, **SKU 62** has the maximum orders.\n",
|
||||
"- Under the **Manufacturing** category, **SKU 17** has the maximum orders.\n",
|
||||
"- Under the **Paper** category, **SKU 107** has the maximum orders.\n",
|
||||
"- Under the **Publishing** category, **SKU 8** has the maximum orders.\n",
|
||||
"- Under the **Utilities** category, **SKU 140** has the maximum orders.\n",
|
||||
"\n",
|
||||
"Given that there are too many ids and only a few records for most of them, consider only the above `Product_ID`s for which there are a maximum number of orders. \n",
|
||||
"\n",
|
||||
"**Note**: The `Invoiced_quantity_in_Pieces` field seems to be a *float* type rather than an *int* type as it should be. This could be because the data itself might be averaged in the first place."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2dbc0d64d157"
|
||||
},
|
||||
"source": [
|
||||
"Check the various prices available for these `Product_ID`s."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "acc1dbd2d838"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_type_food = df[(df[\"Product_ID\"] == \"SKU 62\") & (df[\"Customer_Hierarchy\"] == \"Food\")]\n",
|
||||
"print(\"Food :\")\n",
|
||||
"print(df_type_food[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_manuf = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 17\") & (df[\"Customer_Hierarchy\"] == \"Manufacturing\")\n",
|
||||
"]\n",
|
||||
"print(\"Manufacturing :\")\n",
|
||||
"print(df_type_manuf[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_paper = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 107\") & (df[\"Customer_Hierarchy\"] == \"Paper\")\n",
|
||||
"]\n",
|
||||
"print(\"Paper :\")\n",
|
||||
"print(df_type_paper[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_pub = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 8\") & (df[\"Customer_Hierarchy\"] == \"Publishing\")\n",
|
||||
"]\n",
|
||||
"print(\"Publishing :\")\n",
|
||||
"print(df_type_pub[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_util = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 140\") & (df[\"Customer_Hierarchy\"] == \"Utilities\")\n",
|
||||
"]\n",
|
||||
"print(\"Utilities :\")\n",
|
||||
"print(df_type_util[\"List_Price_Converged\"].value_counts())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f023af578c0f"
|
||||
},
|
||||
"source": [
|
||||
"In the publishing category, `Product_ID` `SKU 8` and `SKU 17` are less than or equal to two different prices in the entire data and so you will exclude them and consider the rest for building the forecast model. The idea here is to train a forecast model on the timeseries data for products with different prices.\n",
|
||||
"\n",
|
||||
"Join the data for all the `Product_ID`s into one dataframe and remove duplicate records."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a44771cc4c20"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_final = pd.concat([df_type_food, df_type_paper, df_type_util])\n",
|
||||
"df_final = (\n",
|
||||
" df_final[\n",
|
||||
" [\n",
|
||||
" \"Product_ID\",\n",
|
||||
" \"Fiscal_Date\",\n",
|
||||
" \"Customer_Hierarchy\",\n",
|
||||
" \"List_Price_Converged\",\n",
|
||||
" \"Invoiced_quantity_in_Pieces\",\n",
|
||||
" ]\n",
|
||||
" ]\n",
|
||||
" .drop_duplicates()\n",
|
||||
" .reset_index(drop=True)\n",
|
||||
")\n",
|
||||
"df_final.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "add5063df368"
|
||||
},
|
||||
"source": [
|
||||
"Save the data to a BigQuery table."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "fd82ba56571f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bq_client = bigquery.Client(project=PROJECT_ID)\n",
|
||||
"\n",
|
||||
"job_config = bigquery.LoadJobConfig(\n",
|
||||
" # Specify a (partial) schema. All columns are always written to the\n",
|
||||
" # table. The schema is used to assist in data type definitions.\n",
|
||||
" schema=[\n",
|
||||
" bigquery.SchemaField(\"Product_ID\", bigquery.enums.SqlTypeNames.STRING),\n",
|
||||
" bigquery.SchemaField(\"Fiscal_Date\", bigquery.enums.SqlTypeNames.DATE),\n",
|
||||
" bigquery.SchemaField(\"List_Price_Converged\", bigquery.enums.SqlTypeNames.FLOAT),\n",
|
||||
" bigquery.SchemaField(\n",
|
||||
" \"Invoiced_quantity_in_Pieces\", bigquery.enums.SqlTypeNames.FLOAT\n",
|
||||
" ),\n",
|
||||
" ],\n",
|
||||
" # Optionally, set the write disposition. BigQuery appends loaded rows\n",
|
||||
" # to an existing table by default, but with WRITE_TRUNCATE write\n",
|
||||
" # disposition it replaces the table with the loaded data.\n",
|
||||
" write_disposition=\"WRITE_TRUNCATE\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# save the dataframe to a table in the created dataset\n",
|
||||
"job = bq_client.load_table_from_dataframe(\n",
|
||||
" df_final,\n",
|
||||
" \"{}.{}.{}\".format(PROJECT_ID, DATASET, TRAINING_DATA_TABLE),\n",
|
||||
" job_config=job_config,\n",
|
||||
") # Make an API request.\n",
|
||||
"job.result() # Wait for the job to complete."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fca77641b03b"
|
||||
},
|
||||
"source": [
|
||||
"# Train the model using BigQuery ML\n",
|
||||
"<a name=\"section-9\"></a>\n",
|
||||
"\n",
|
||||
"Train an [Arima-Plus](https://cloud.google.com/bigquery-ml/docs/reference/standard-sql/bigqueryml-syntax-create-time-series) model on the data using BigQuery ML."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cded27507891"
|
||||
},
|
||||
"source": [
|
||||
"#@bigquery\n",
|
||||
"create or replace model pricing_optimization.bqml_arima\n",
|
||||
"options\n",
|
||||
" (model_type = 'ARIMA_PLUS',\n",
|
||||
" time_series_timestamp_col = 'Fiscal_Date',\n",
|
||||
" time_series_data_col = 'Invoiced_quantity_in_Pieces',\n",
|
||||
" time_series_id_col = 'ID'\n",
|
||||
" ) as\n",
|
||||
"select\n",
|
||||
" Fiscal_Date,\n",
|
||||
" Concat(Product_ID,\"_\" ,Cast(List_Price_Converged as string)) as ID,\n",
|
||||
" Invoiced_quantity_in_Pieces\n",
|
||||
"from\n",
|
||||
" pricing_optimization.TRAINING_DATA\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "332fd11ff32b"
|
||||
},
|
||||
"source": [
|
||||
"## Generate forecasts from the model\n",
|
||||
"<a name=\"section-10\"></a>\n",
|
||||
"\n",
|
||||
"Predict the sales for the next 30 days for each id and save to a dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ef926cdbf28e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"client = Client()\n",
|
||||
"\n",
|
||||
"query = '''\n",
|
||||
"DECLARE HORIZON STRING DEFAULT \"30\"; #number of values to forecast\n",
|
||||
"DECLARE CONFIDENCE_LEVEL STRING DEFAULT \"0.90\"; ## required confidence level\n",
|
||||
"\n",
|
||||
"EXECUTE IMMEDIATE format(\"\"\"\n",
|
||||
" SELECT\n",
|
||||
" *\n",
|
||||
" FROM \n",
|
||||
" ML.FORECAST(MODEL pricing_optimization.bqml_arima, \n",
|
||||
" STRUCT(%s AS horizon, \n",
|
||||
" %s AS confidence_level)\n",
|
||||
" )\n",
|
||||
" \"\"\",HORIZON,CONFIDENCE_LEVEL)'''\n",
|
||||
"job = client.query(query)\n",
|
||||
"dfforecast = job.to_dataframe()\n",
|
||||
"dfforecast.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "608c7de72dae"
|
||||
},
|
||||
"source": [
|
||||
"## Interpret the results to choose the best price\n",
|
||||
"<a name=\"section-11\"></a>\n",
|
||||
"\n",
|
||||
"Calculate average forecast values for the forecast duration."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e1e193680400"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dfforecast_avg = (\n",
|
||||
" dfforecast[[\"ID\", \"forecast_value\"]].groupby(\"ID\", as_index=False).mean()\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5ce395d652a3"
|
||||
},
|
||||
"source": [
|
||||
"Extract the ID and Price fields from the ID field."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "452c56fa58ed"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dfforecast_avg[\"Product_ID\"] = dfforecast_avg[\"ID\"].apply(lambda x: x.split(\"_\")[0])\n",
|
||||
"dfforecast_avg[\"Price\"] = dfforecast_avg[\"ID\"].apply(lambda x: x.split(\"_\")[1])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3cee67f4028f"
|
||||
},
|
||||
"source": [
|
||||
"Plot the average forecasted sales vs. the price of the product."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "fb351c8f383d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in dfforecast_avg[\"Product_ID\"].unique():\n",
|
||||
" dfforecast_avg[dfforecast_avg[\"Product_ID\"] == i].set_index(\"Price\").sort_values(\n",
|
||||
" \"forecast_value\"\n",
|
||||
" ).plot(kind=\"bar\")\n",
|
||||
" plt.title(\"Price vs. Average Sales for \" + i)\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "67ff3acc74a5"
|
||||
},
|
||||
"source": [
|
||||
"Based on the plots for price vs. the average forecasted orders, it can be said that to use the maximum orders, each of the considered `Product_ID`s can follow the below prices:\n",
|
||||
"\n",
|
||||
"- SKU 107's price range can be from 4.44 - 4.73 units\n",
|
||||
"- SKU 140's price can be 1.95 units\n",
|
||||
"- SKU 62's price can be 4.23 units\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Clean Up\n",
|
||||
"<a name=\"section-12\"></a>\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial. The following code deletes the entire dataset."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "d78908b8134d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Construct a BigQuery client object.\n",
|
||||
"client = bigquery.Client()\n",
|
||||
"\n",
|
||||
"# TODO(developer): Set model_id to the ID of the model to fetch.\n",
|
||||
"dataset_id = \"{PROJECT}.{DATASET}\".format(PROJECT=PROJECT_ID, DATASET=DATASET)\n",
|
||||
"\n",
|
||||
"# Use the delete_contents parameter to delete a dataset and its contents.\n",
|
||||
"# Use the not_found_ok parameter to not receive an error if the dataset has already been deleted.\n",
|
||||
"client.delete_dataset(\n",
|
||||
" dataset_id, delete_contents=True, not_found_ok=True\n",
|
||||
") # Make an API request.\n",
|
||||
"\n",
|
||||
"print(\"Deleted dataset '{}'.\".format(dataset_id))"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"name": "pricing-optimization.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -459,7 +459,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil cp gs://cloud-samples-data/ai-platform-unified/matching_engine/glove-100-angular.hdf5 ."
|
||||
"! gsutil cp gs://cloud-samples-data/vertex-ai/matching_engine/glove-100-angular.hdf5 ."
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -1,877 +0,0 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ur8xi4C7S06n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
"# You may obtain a copy of the License at\n",
|
||||
"#\n",
|
||||
"# https://www.apache.org/licenses/LICENSE-2.0\n",
|
||||
"#\n",
|
||||
"# Unless required by applicable law or agreed to in writing, software\n",
|
||||
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
|
||||
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
|
||||
"# See the License for the specific language governing permissions and\n",
|
||||
"# limitations under the License."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "JAPoU8Sm5E6e"
|
||||
},
|
||||
"source": [
|
||||
"<table align=\"left\">\n",
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "WBFL9LagqmwT"
|
||||
},
|
||||
"source": [
|
||||
"#Vertex AI: Track parameters and metrics for locally trained models"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "tvgnzT1CKxrO"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"This notebook demonstrates how to track metrics and parameters for ML training jobs and analyze this metadata using Vertex SDK for Python.\n",
|
||||
"\n",
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"In this notebook, we will train a simple distributed neural network (DNN) model to predict automobile's miles per gallon (MPG) based on automobile information in the [auto-mpg dataset](https://www.kaggle.com/devanshbesain/exploration-and-analysis-auto-mpg).\n",
|
||||
"\n",
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this notebook, you will learn how to use Vertex SDK for Python to:\n",
|
||||
"\n",
|
||||
" * Track parameters and metrics for a locally trainined model.\n",
|
||||
" * Extract and perform analysis for all parameters and metrics within an Experiment.\n",
|
||||
"\n",
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "ze4-nDLfK4pw"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your local development environment\n",
|
||||
"\n",
|
||||
"**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n",
|
||||
"all the requirements to run this notebook. You can skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "gCuSR8GkAgzl"
|
||||
},
|
||||
"source": [
|
||||
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
|
||||
"You need the following:\n",
|
||||
"\n",
|
||||
"* The Google Cloud SDK\n",
|
||||
"* Git\n",
|
||||
"* Python 3\n",
|
||||
"* virtualenv\n",
|
||||
"* Jupyter notebook running in a virtual environment with Python 3\n",
|
||||
"\n",
|
||||
"The Google Cloud guide to [Setting up a Python development\n",
|
||||
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
|
||||
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
|
||||
"for meeting these requirements. The following steps provide a condensed set of\n",
|
||||
"instructions:\n",
|
||||
"\n",
|
||||
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
|
||||
"\n",
|
||||
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
|
||||
"\n",
|
||||
"1. [Install\n",
|
||||
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
|
||||
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
|
||||
"\n",
|
||||
"1. To install Jupyter, run `pip install jupyter` on the\n",
|
||||
"command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"1. Open this notebook in the Jupyter Notebook Dashboard."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "i7EUnXsZhAGF"
|
||||
},
|
||||
"source": [
|
||||
"### Install additional packages\n",
|
||||
"\n",
|
||||
"Run the following commands to install the Vertex SDK for Python."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "IaYsrh0Tc17L"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"if \"google.colab\" in sys.modules:\n",
|
||||
" USER_FLAG = \"\"\n",
|
||||
"else:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "wyy5Lbnzg5fi"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!python3 -m pip install {USER_FLAG} google-cloud-aiplatform --upgrade"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "hhq5zEbGg0XX"
|
||||
},
|
||||
"source": [
|
||||
"### Restart the kernel\n",
|
||||
"\n",
|
||||
"After you install the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "EzrelQZ22IZj"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Automatically restart kernel after installs\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Automatically restart kernel after installs\n",
|
||||
" import IPython\n",
|
||||
"\n",
|
||||
" app = IPython.Application.instance()\n",
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "lWEdiXsJg0XY"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Select a GPU runtime\n",
|
||||
"\n",
|
||||
"**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "BF1j6f9HApxa"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "WReHDGG5g0XY"
|
||||
},
|
||||
"source": [
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "oM1iC_MfAts1"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "qJYoRfYng0XZ"
|
||||
},
|
||||
"source": [
|
||||
"Otherwise, set your project ID here."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "riG_qUokg0XZ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
|
||||
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "06571eb4063b"
|
||||
},
|
||||
"source": [
|
||||
"#### Timestamp\n",
|
||||
"\n",
|
||||
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "697568e92bd6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "dr--iN2kAylZ"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already\n",
|
||||
"authenticated. Skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "sBCra4QMA2wR"
|
||||
},
|
||||
"source": [
|
||||
"**If you are using Colab**, run the cell below and follow the instructions\n",
|
||||
"when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"1. In the Cloud Console, go to the [**Create service account key**\n",
|
||||
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
|
||||
"\n",
|
||||
"2. Click **Create service account**.\n",
|
||||
"\n",
|
||||
"3. In the **Service account name** field, enter a name, and\n",
|
||||
" click **Create**.\n",
|
||||
"\n",
|
||||
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
|
||||
"into the filter box, and select\n",
|
||||
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
|
||||
"local environment.\n",
|
||||
"\n",
|
||||
"6. Enter the path to your service account key as the\n",
|
||||
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "PyQmSRbKA8r-"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebooks, then don't execute this code\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "XoEqT2Y4DJmf"
|
||||
},
|
||||
"source": [
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "Y9Uo3tifg1kx"
|
||||
},
|
||||
"source": [
|
||||
"Import required libraries."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "pRUOFELefqf1"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import pandas as pd\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from tensorflow.python.keras import Sequential, layers\n",
|
||||
"from tensorflow.python.keras.utils import data_utils"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "xtXZWmYqJ1bh"
|
||||
},
|
||||
"source": [
|
||||
"Define some constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "JIOrI-hoJ46P"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPERIMENT_NAME = \"\" # @param {type:\"string\"}\n",
|
||||
"REGION = \"[your-region]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "jWQLXXNVN4Lv"
|
||||
},
|
||||
"source": [
|
||||
"If EXEPERIMENT_NAME is not set, set a default one below:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Q1QInYWOKsmo"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if EXPERIMENT_NAME == \"\" or EXPERIMENT_NAME is None:\n",
|
||||
" EXPERIMENT_NAME = \"my-experiment-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "Xuny18aMcWDb"
|
||||
},
|
||||
"source": [
|
||||
"## Concepts\n",
|
||||
"\n",
|
||||
"To better understanding how parameters and metrics are stored and organized, we'd like to introduce the following concepts:\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "NThDci5bp0Uw"
|
||||
},
|
||||
"source": [
|
||||
"### Experiment\n",
|
||||
"Experiments describe a context that groups your runs and the artifacts you create into a logical session. For example, in this notebook you create an Experiment and log data to that experiment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "SAyRR3Ydp4X5"
|
||||
},
|
||||
"source": [
|
||||
"### Run\n",
|
||||
"A run represents a single path/avenue that you executed while performing an experiment. A run includes artifacts that you used as inputs or outputs, and parameters that you used in this execution. An Experiment can contain multiple runs. "
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "l1YW2pgyegFP"
|
||||
},
|
||||
"source": [
|
||||
"## Getting started tracking parameters and metrics\n",
|
||||
"\n",
|
||||
"You can use the Vertex SDK for Python to track metrics and parameters for models trained locally. \n",
|
||||
"\n",
|
||||
"In the following example, you train a simple distributed neural network (DNN) model to predict automobile's miles per gallon (MPG) based on automobile information in the [auto-mpg dataset](https://www.kaggle.com/devanshbesain/exploration-and-analysis-auto-mpg)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "KPY41M9_AhZU"
|
||||
},
|
||||
"source": [
|
||||
"### Load and process the training dataset"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bfMQSmRuUuX-"
|
||||
},
|
||||
"source": [
|
||||
"Download and process the dataset."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "RiQuMv4bmpuV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def read_data(uri):\n",
|
||||
" dataset_path = data_utils.get_file(\"auto-mpg.data\", uri)\n",
|
||||
" column_names = [\n",
|
||||
" \"MPG\",\n",
|
||||
" \"Cylinders\",\n",
|
||||
" \"Displacement\",\n",
|
||||
" \"Horsepower\",\n",
|
||||
" \"Weight\",\n",
|
||||
" \"Acceleration\",\n",
|
||||
" \"Model Year\",\n",
|
||||
" \"Origin\",\n",
|
||||
" ]\n",
|
||||
" raw_dataset = pd.read_csv(\n",
|
||||
" dataset_path,\n",
|
||||
" names=column_names,\n",
|
||||
" na_values=\"?\",\n",
|
||||
" comment=\"\\t\",\n",
|
||||
" sep=\" \",\n",
|
||||
" skipinitialspace=True,\n",
|
||||
" )\n",
|
||||
" dataset = raw_dataset.dropna()\n",
|
||||
" dataset[\"Origin\"] = dataset[\"Origin\"].map(\n",
|
||||
" lambda x: {1: \"USA\", 2: \"Europe\", 3: \"Japan\"}.get(x)\n",
|
||||
" )\n",
|
||||
" dataset = pd.get_dummies(dataset, prefix=\"\", prefix_sep=\"\")\n",
|
||||
" return dataset\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"dataset = read_data(\n",
|
||||
" \"http://archive.ics.uci.edu/ml/machine-learning-databases/auto-mpg/auto-mpg.data\"\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "Y06J7A7yU21t"
|
||||
},
|
||||
"source": [
|
||||
"Split dataset for training and testing."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "p5JBCBKyH-NC"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def train_test_split(dataset, split_frac=0.8, random_state=0):\n",
|
||||
" train_dataset = dataset.sample(frac=split_frac, random_state=random_state)\n",
|
||||
" test_dataset = dataset.drop(train_dataset.index)\n",
|
||||
" train_labels = train_dataset.pop(\"MPG\")\n",
|
||||
" test_labels = test_dataset.pop(\"MPG\")\n",
|
||||
"\n",
|
||||
" return train_dataset, test_dataset, train_labels, test_labels\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"train_dataset, test_dataset, train_labels, test_labels = train_test_split(dataset)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "gaNNTFPaU7KT"
|
||||
},
|
||||
"source": [
|
||||
"Normalize the features in the dataset for better model performance."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "VGq5QCoyIEWJ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def normalize_dataset(train_dataset, test_dataset):\n",
|
||||
" train_stats = train_dataset.describe()\n",
|
||||
" train_stats = train_stats.transpose()\n",
|
||||
"\n",
|
||||
" def norm(x):\n",
|
||||
" return (x - train_stats[\"mean\"]) / train_stats[\"std\"]\n",
|
||||
"\n",
|
||||
" normed_train_data = norm(train_dataset)\n",
|
||||
" normed_test_data = norm(test_dataset)\n",
|
||||
"\n",
|
||||
" return normed_train_data, normed_test_data\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"normed_train_data, normed_test_data = normalize_dataset(train_dataset, test_dataset)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "UBXUgxgqA_GB"
|
||||
},
|
||||
"source": [
|
||||
"### Define ML model and training function"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "66odBYKrIN4q"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def train(\n",
|
||||
" train_data,\n",
|
||||
" train_labels,\n",
|
||||
" num_units=64,\n",
|
||||
" activation=\"relu\",\n",
|
||||
" dropout_rate=0.0,\n",
|
||||
" validation_split=0.2,\n",
|
||||
" epochs=1000,\n",
|
||||
"):\n",
|
||||
"\n",
|
||||
" model = Sequential(\n",
|
||||
" [\n",
|
||||
" layers.Dense(\n",
|
||||
" num_units,\n",
|
||||
" activation=activation,\n",
|
||||
" input_shape=[len(train_dataset.keys())],\n",
|
||||
" ),\n",
|
||||
" layers.Dropout(rate=dropout_rate),\n",
|
||||
" layers.Dense(num_units, activation=activation),\n",
|
||||
" layers.Dense(1),\n",
|
||||
" ]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" model.compile(loss=\"mse\", optimizer=\"adam\", metrics=[\"mae\", \"mse\"])\n",
|
||||
" print(model.summary())\n",
|
||||
"\n",
|
||||
" history = model.fit(\n",
|
||||
" train_data, train_labels, epochs=epochs, validation_split=validation_split\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" return model, history"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "O8XJZB3gR8eL"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize the Vertex AI SDK for Python and create an Experiment\n",
|
||||
"\n",
|
||||
"Initialize the *client* for Vertex AI and create an experiment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "o_wnT10RJ7-W"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, experiment=EXPERIMENT_NAME)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "u-iTnzt3B6Z_"
|
||||
},
|
||||
"source": [
|
||||
"### Start several model training runs\n",
|
||||
"\n",
|
||||
"Training parameters and metrics are logged for each run."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "i2wnpu8_7JfV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"parameters = [\n",
|
||||
" {\"num_units\": 16, \"epochs\": 3, \"dropout_rate\": 0.1},\n",
|
||||
" {\"num_units\": 16, \"epochs\": 10, \"dropout_rate\": 0.1},\n",
|
||||
" {\"num_units\": 16, \"epochs\": 10, \"dropout_rate\": 0.2},\n",
|
||||
" {\"num_units\": 32, \"epochs\": 10, \"dropout_rate\": 0.1},\n",
|
||||
" {\"num_units\": 32, \"epochs\": 10, \"dropout_rate\": 0.2},\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"for i, params in enumerate(parameters):\n",
|
||||
" aiplatform.start_run(run=f\"auto-mpg-local-run-{i}\")\n",
|
||||
" aiplatform.log_params(params)\n",
|
||||
" model, history = train(\n",
|
||||
" normed_train_data,\n",
|
||||
" train_labels,\n",
|
||||
" num_units=params[\"num_units\"],\n",
|
||||
" activation=\"relu\",\n",
|
||||
" epochs=params[\"epochs\"],\n",
|
||||
" dropout_rate=params[\"dropout_rate\"],\n",
|
||||
" )\n",
|
||||
" aiplatform.log_metrics(\n",
|
||||
" {metric: values[-1] for metric, values in history.history.items()}\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" loss, mae, mse = model.evaluate(normed_test_data, test_labels, verbose=2)\n",
|
||||
" aiplatform.log_metrics({\"eval_loss\": loss, \"eval_mae\": mae, \"eval_mse\": mse})"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "jZLrJZTfL7tE"
|
||||
},
|
||||
"source": [
|
||||
"### Extract parameters and metrics into a dataframe for analysis"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "A1PqKxlpOZa2"
|
||||
},
|
||||
"source": [
|
||||
"We can also extract all parameters and metrics associated with any Experiment into a dataframe for further analysis."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "jbRf1WoH_vbY"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"experiment_df = aiplatform.get_experiment_df()\n",
|
||||
"experiment_df"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "EYuYgqVCMKU1"
|
||||
},
|
||||
"source": [
|
||||
"### Visualizing an experiment's parameters and metrics"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "r8orCj8iJuO1"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"plt.rcParams[\"figure.figsize\"] = [15, 5]\n",
|
||||
"\n",
|
||||
"ax = pd.plotting.parallel_coordinates(\n",
|
||||
" experiment_df.reset_index(level=0),\n",
|
||||
" \"run_name\",\n",
|
||||
" cols=[\n",
|
||||
" \"param.num_units\",\n",
|
||||
" \"param.dropout_rate\",\n",
|
||||
" \"param.epochs\",\n",
|
||||
" \"metric.loss\",\n",
|
||||
" \"metric.val_loss\",\n",
|
||||
" \"metric.eval_loss\",\n",
|
||||
" ],\n",
|
||||
" color=[\"blue\", \"green\", \"pink\", \"red\"],\n",
|
||||
")\n",
|
||||
"ax.set_yscale(\"symlog\")\n",
|
||||
"ax.legend(bbox_to_anchor=(1.0, 0.5))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "WTHvPMweMlP1"
|
||||
},
|
||||
"source": [
|
||||
"## Visualizing experiments in Cloud Console"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "F19_5lw0MqXv"
|
||||
},
|
||||
"source": [
|
||||
"Run the following to get the URL of Vertex AI Experiments for your project.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "GmN9vE9pqqzt"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"print(\"Vertex AI Experiments:\")\n",
|
||||
"print(\n",
|
||||
" f\"https://console.cloud.google.com/ai/platform/experiments/experiments?folder=&organizationId=&project={PROJECT_ID}\"\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "TpV-iwP9qw9c"
|
||||
},
|
||||
"source": [
|
||||
"## Cleaning up\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "sdk-metric-parameter-tracking-for-locally-trained-models.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -11,8 +11,8 @@ The purpose of this set of notebooks and markdown files is to demonstrate Google
|
||||
1. [Data Management](stage1)
|
||||
2. [Experimentation](stage2)
|
||||
3. [Formalization](stage3)
|
||||
4. Evaluation
|
||||
5. Deployment
|
||||
6. Serving
|
||||
4. [Evaluation](stage4)
|
||||
5. [Deployment](stage5)
|
||||
6. [Serving](stage6)
|
||||
7. Monitoring
|
||||
8. Continuous Training
|
||||
|
||||
@@ -22,7 +22,7 @@ The first stage in MLOps is the collection and preparation for the purpose of de
|
||||
- Data is preprocessed for training and evaluation using Dataflow.
|
||||
- Data augmentation is performed on-the-fly and is coupled with model feeding.
|
||||
|
||||
<img src='stage1.jpg'>
|
||||
<img src='stage1v2.png'>
|
||||
|
||||
## Notebooks
|
||||
|
||||
@@ -30,10 +30,78 @@ The first stage in MLOps is the collection and preparation for the purpose of de
|
||||
|
||||
[Get Started with BQ datasets](get_started_bq_datasets.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Vertex AI `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.
|
||||
- Extract a copy of the dataset from `BigQuery` to a CSV file in Cloud Storage -- compatible for `AutoML` or custom training.
|
||||
- Select rows from a `BigQuery` dataset into a `pandas` dataframe -- compatible for custom training.
|
||||
- Select rows from a `BigQuery` dataset into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.
|
||||
- Select rows from extracted CSV files into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.
|
||||
- Create a `BigQuery` dataset from CSV files.
|
||||
- Extract data from `BigQuery` table into a `DMatrix` -- compatible for custom training `XGBoost` models.
|
||||
```
|
||||
|
||||
[Get Started with Vertex datasets](get_started_vertex_datasets.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Vertex AI `Dataset` resource for:
|
||||
- image data
|
||||
- text data
|
||||
- video data
|
||||
- tabular data
|
||||
- forecasting data
|
||||
|
||||
|
||||
- Search `Dataset` resources using a filter.
|
||||
- Read a sample of a `BigQuery` dataset into a dataframe.
|
||||
- Generate statistics and data schema using TensorFlow Data Validation from the samples in the dataframe.
|
||||
- Detect anomalies in new data using TensorFlow Data Validation.
|
||||
- Generate a TFRecord feature specification using TensorFlow Transform from the data schema.
|
||||
- Export a dataset and convert to TFRecords.
|
||||
```
|
||||
|
||||
[Get Started with Dataflow](get_started_dataflow.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Offline preprocessing of data:
|
||||
- Serially - w/o dataflow
|
||||
- Parallel - with dataflow
|
||||
- Upstream preprocessing of data:
|
||||
- tabular data
|
||||
- image data
|
||||
```
|
||||
|
||||
[Get Started with Data Labeling](get_started_with_data_labeling.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Specialist Pool for data labelers.
|
||||
- Create a data labeling job.
|
||||
- Submit the data labeling job.
|
||||
- List data labeling jobs.
|
||||
- Cancel a data labeling job.
|
||||
```
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 1: Data Management](mlops_data_management.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Explore and visualize the data.
|
||||
- Create a Vertex AI `Dataset` resource from `BigQuery` table -- for AutoML training.
|
||||
- Extract a copy of the dataset to a CSV file in Cloud Storage.
|
||||
- Create a Vertex AI `Dataset` resource from CSV files -- alternative for AutoML training.
|
||||
- Read a sample of the `BigQuery` dataset into a dataframe.
|
||||
- Generate statistics and data schema using TensorFlow Data Validation from the samples in the dataframe.
|
||||
- Generate a TFRecord feature specification using TensorFlow Data Validation from the data schema.
|
||||
- Preprocess a portion of the BigQuery data using `Dataflow` -- for custom training.
|
||||
```
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2020 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,14 +33,19 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -67,7 +72,7 @@
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
"The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). In this version of the dataset you consider the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -82,12 +87,12 @@
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex Datasets`\n",
|
||||
"- `Vertex AI Datasets`\n",
|
||||
"- `BigQuery Datasets`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Create a Vertex `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.\n",
|
||||
"- Create a Vertex AI `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.\n",
|
||||
"- Extract a copy of the dataset from `BigQuery` to a CSV file in Cloud Storage -- compatible for `AutoML` or custom training.\n",
|
||||
"- Select rows from a `BigQuery` dataset into a `pandas` dataframe -- compatible for custom training.\n",
|
||||
"- Select rows from a `BigQuery` dataset into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.\n",
|
||||
@@ -104,10 +109,10 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with structured (tabular) data in BigQuery:\n",
|
||||
"When doing E2E MLOps on Google Cloud, following are the best practices when dealing with structured (tabular) data in BigQuery:\n",
|
||||
"\n",
|
||||
"- For AutoML training:\n",
|
||||
" - Create a managed dataset with Vertex `TabularDataset`.\n",
|
||||
" - Create a managed dataset with Vertex AI `TabularDataset`.\n",
|
||||
" - Use the BigQuery table as the input to the dataset.\n",
|
||||
" - Specify columns and columns transformations when running the AutoML training pipeline job.\n",
|
||||
"\n",
|
||||
@@ -124,7 +129,7 @@
|
||||
" - Within the generator (upstream)\n",
|
||||
" - Within the model (downstream)\n",
|
||||
" - XGBoost model training:\n",
|
||||
" - Use BigQuery ML builtin XGBoost training.\n",
|
||||
" - Use BigQuery ML built-in XGBoost training.\n",
|
||||
" - Alternatively, create a DMatrix generator from CSV files extracted from BigQuery table.\n",
|
||||
" - Pytorch model training:\n",
|
||||
" - Extract the BigQuery to a pandas dataframe.\n",
|
||||
@@ -132,10 +137,19 @@
|
||||
" - Create a DataLoader generator from the pandas dataframe.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"- Alternately:\n",
|
||||
"- Alternatively:\n",
|
||||
" - Extract the BigQuery table to CSV files.\n",
|
||||
" - Preprocess the CSV files.\n",
|
||||
" - Create a tf.data.Dataset generator from the CSV files."
|
||||
" - Create a tf.data.Dataset generator from the CSV files.\n",
|
||||
" \n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"- BigQuery\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -146,7 +160,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages to execute this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -157,38 +171,26 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "install_xgboost"
|
||||
},
|
||||
"source": [
|
||||
"Install the latest GA version of *XGBoost* library as well."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "install_xgboost"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install -U xgboost $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"# Install the packages\n",
|
||||
"! pip3 install --upgrade pyarrow $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-bigquery $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q\n",
|
||||
"! pip3 install -U xgboost $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-io==0.18 $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -220,6 +222,32 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "84cd83853240"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, BigQuery, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,bigquery,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -285,7 +313,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -296,7 +324,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -323,6 +354,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "77c385f0db59"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "535223fa4b84"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -333,12 +425,7 @@
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"When you submit a custom training job using the Vertex SDK, you upload a Python package\n",
|
||||
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
|
||||
"the code from this package. In this tutorial, Vertex AI also saves the\n",
|
||||
"trained model that results from your job in the same bucket. You can then\n",
|
||||
"create an `Endpoint` resource based on this output in order to serve\n",
|
||||
"online predictions.\n",
|
||||
"When you create a dataset resource using the Vertex SDK, you can provide a Cloud Storage bucket that contains the data. Vertex AI creates the dataset resource from the data. In this tutorial, Vertex AI also creates a dataset resource from your data in the Cloud Storage bucket.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
|
||||
]
|
||||
@@ -351,7 +438,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -362,8 +450,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -383,7 +472,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -403,7 +492,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -412,9 +501,6 @@
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
@@ -426,84 +512,21 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"source": [
|
||||
"#### Import BigQuery\n",
|
||||
"\n",
|
||||
"Import the BigQuery package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aiplatform\n",
|
||||
"import pandas as pd\n",
|
||||
"import xgboost as xgb\n",
|
||||
"from google.cloud import bigquery"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_xgboost"
|
||||
},
|
||||
"source": [
|
||||
"#### Import XGBoost\n",
|
||||
"\n",
|
||||
"Import the XGBoost package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_xgboost"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import xgboost as xgb"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_pandas"
|
||||
},
|
||||
"source": [
|
||||
"#### Import pandas\n",
|
||||
"\n",
|
||||
"Import the pandas package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_pandas"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pandas as pd"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,region"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex SDK for Python\n",
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -514,7 +537,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION)"
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -536,7 +559,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bqclient = bigquery.Client()"
|
||||
"bqclient = bigquery.Client(project=PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -547,7 +570,7 @@
|
||||
"source": [
|
||||
"#### Location of BigQuery training data.\n",
|
||||
"\n",
|
||||
"Now set the variable `IMPORT_FILE` to the location of the data table in BigQuery."
|
||||
"Now, set the variable `IMPORT_FILE` to the location of the data table in BigQuery and `BQ_TABLE` with the table id."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -589,10 +612,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
"dataset = aiplatform.TabularDataset.create(\n",
|
||||
" display_name=\"NOAA historical weather data\" + \"_\" + TIMESTAMP,\n",
|
||||
" bq_source=[IMPORT_FILE],\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_URI[5:]},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"label_column = \"mean_temp\"\n",
|
||||
@@ -608,7 +631,7 @@
|
||||
"source": [
|
||||
"### Copy the dataset to Cloud Storage\n",
|
||||
"\n",
|
||||
"Next, you make a copy of the BigQuery dataset, as a CSV file, to Cloud Storage using the BigQuery extract command.\n",
|
||||
"Next, you make a copy of the BigQuery table as a CSV file, to Cloud Storage using the BigQuery extract command.\n",
|
||||
"\n",
|
||||
"Learn more about [BigQuery command line interface](https://cloud.google.com/bigquery/docs/reference/bq-cli-reference)."
|
||||
]
|
||||
@@ -624,9 +647,9 @@
|
||||
"comps = BQ_TABLE.split(\".\")\n",
|
||||
"BQ_PROJECT_DATASET_TABLE = comps[0] + \":\" + comps[1] + \".\" + comps[2]\n",
|
||||
"\n",
|
||||
"! bq --location=us extract --destination_format CSV $BQ_PROJECT_DATASET_TABLE $BUCKET_NAME/mydata*.csv\n",
|
||||
"! bq --location=us extract --destination_format CSV $BQ_PROJECT_DATASET_TABLE $BUCKET_URI/mydata*.csv\n",
|
||||
"\n",
|
||||
"IMPORT_FILES = ! gsutil ls $BUCKET_NAME/mydata*.csv\n",
|
||||
"IMPORT_FILES = ! gsutil ls $BUCKET_URI/mydata*.csv\n",
|
||||
"\n",
|
||||
"print(IMPORT_FILES)\n",
|
||||
"\n",
|
||||
@@ -662,15 +685,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" gcs_source = IMPORT_FILES\n",
|
||||
"else:\n",
|
||||
" gcs_source = [IMPORT_FILE]\n",
|
||||
"gcs_source = IMPORT_FILES\n",
|
||||
"\n",
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
"dataset = aiplatform.TabularDataset.create(\n",
|
||||
" display_name=\"NOAA historical weather data\" + \"_\" + TIMESTAMP,\n",
|
||||
" gcs_source=gcs_source,\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_URI[5:]},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -692,6 +712,30 @@
|
||||
"Learn more about [Creating BigQuery views](https://cloud.google.com/bigquery/docs/views)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "7dc142433e50"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Set dataset name and view name in BigQuery\n",
|
||||
"BQ_MY_DATASET = \"[your-dataset-name]\"\n",
|
||||
"BQ_MY_TABLE = \"[your-view-name]\"\n",
|
||||
"\n",
|
||||
"# Otherwise, use the default names\n",
|
||||
"if (\n",
|
||||
" BQ_MY_DATASET == \"\"\n",
|
||||
" or BQ_MY_DATASET is None\n",
|
||||
" or BQ_MY_DATASET == \"[your-dataset-name]\"\n",
|
||||
"):\n",
|
||||
" BQ_MY_DATASET = \"mlops_dataset_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"if BQ_MY_TABLE == \"\" or BQ_MY_TABLE is None or BQ_MY_TABLE == \"[your-view-name]\":\n",
|
||||
" BQ_MY_TABLE = \"mlops_view_\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -700,8 +744,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BQ_MY_DATASET = 'mydataset'\n",
|
||||
"BQ_MY_TABLE = 'myview'\n",
|
||||
"# Create the resources\n",
|
||||
"! bq --location=US mk -d \\\n",
|
||||
"$PROJECT_ID:$BQ_MY_DATASET\n",
|
||||
"\n",
|
||||
@@ -742,8 +785,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download a table.\n",
|
||||
"table = bigquery.TableReference.from_string(\"bigquery-public-data.samples.gsod\")\n",
|
||||
"# Download the table.\n",
|
||||
"table = bigquery.TableReference.from_string(BQ_TABLE)\n",
|
||||
"\n",
|
||||
"rows = bqclient.list_rows(\n",
|
||||
" table,\n",
|
||||
@@ -1029,22 +1072,6 @@
|
||||
"TABLE_ID = \"gsod\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_bigquery_dataset(dataset_id):\n",
|
||||
" dataset = bigquery.Dataset(\n",
|
||||
" bigquery.dataset.DatasetReference(PROJECT_ID, dataset_id)\n",
|
||||
" )\n",
|
||||
" dataset.location = \"us\"\n",
|
||||
"\n",
|
||||
" try:\n",
|
||||
" dataset = bqclient.create_dataset(dataset) # API request\n",
|
||||
" return True\n",
|
||||
" except Exception as err:\n",
|
||||
" print(err)\n",
|
||||
" if err.code != 409: # http_client.CONFLICT\n",
|
||||
" raise\n",
|
||||
" return False\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def load_data_into_bigquery(url, dataset_id, table_id):\n",
|
||||
" create_bigquery_dataset(dataset_id)\n",
|
||||
" dataset = bqclient.dataset(dataset_id)\n",
|
||||
@@ -1077,13 +1104,11 @@
|
||||
"source": [
|
||||
"### Read BigQuery table into XGboost DMatrix\n",
|
||||
"\n",
|
||||
"Currently, there is no direct data feeding connector between BigQuery and the open source XGBoost.\n",
|
||||
"Currently, there is no direct data feeding connector between BigQuery and the open source XGBoost. The BigQuery ML service has a built-in XGBoost training module.\n",
|
||||
"\n",
|
||||
"The BigQuery ML service has XGBoost training builtin.\n",
|
||||
"Alernatively, you extract the data either as a pandas dataframe or as CSV files. The extracted data is then given as an input to a `DMatrix` object when training the model.\n",
|
||||
"\n",
|
||||
"Alernatively, you extract the data either as a pandas dataframe or as CSV files. The extracted data is then inputted to a `DMatrix` object when training the model.\n",
|
||||
"\n",
|
||||
"Learn more about [Getting started with builtin XGBoost](https://cloud.google.com/ai-platform/training/docs/algorithms/xgboost-start)"
|
||||
"Learn more about [Getting started with built-in XGBoost](https://cloud.google.com/ai-platform/training/docs/algorithms/xgboost-start)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1094,7 +1119,7 @@
|
||||
"source": [
|
||||
"### Read pandas table into XGboost DMatrix\n",
|
||||
"\n",
|
||||
"Next, you load the pandas dataframe into a `DMatrix` object. XGBoost does not support non-numeric inputs. Any column that is categorical will need to be one-hot encoded prior to loading the dataframe."
|
||||
"Next, you load the pandas dataframe into a `DMatrix` object. XGBoost does not support non-numeric inputs. Any column that is categorical need to be one-hot encoded prior to loading the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1107,7 +1132,7 @@
|
||||
"source": [
|
||||
"dataframe[\"station_number\"] = pd.to_numeric(dataframe[\"station_number\"])\n",
|
||||
"labels = dataframe[\"mean_temp\"]\n",
|
||||
"data = dataframe.drop(4)\n",
|
||||
"data = dataframe.drop([\"mean_temp\"], axis=1)\n",
|
||||
"\n",
|
||||
"dtrain = xgb.DMatrix(data, label=labels)"
|
||||
]
|
||||
@@ -1120,7 +1145,7 @@
|
||||
"source": [
|
||||
"### Read CSV files into XGboost DMatrix\n",
|
||||
"\n",
|
||||
"Currently, there is no Cloud Storage support in XGBoost. If you use CSV files for input, you will need to download them locally."
|
||||
"Currently, there is no Cloud Storage support in XGBoost. If you use CSV files for input, you need to download them locally."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1142,86 +1167,42 @@
|
||||
"id": "cleanup:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"# Cleaning up\n",
|
||||
"# Clean up\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"- Vertex AI Dataset resource\n",
|
||||
"- Cloud Storage Bucket\n",
|
||||
"- BigQuery Dataset\n",
|
||||
"\n",
|
||||
"Set `delete_storage` to _True_ to delete the storage resources used in this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "cleanup:mbsdk"
|
||||
"id": "47ad926d84e8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the dataset using the Vertex dataset object\n",
|
||||
"dataset.delete()\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the temporary BigQuery dataset\n",
|
||||
"! bq rm -r -f $PROJECT_ID:$DATASET_ID\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"delete_storage = False\n",
|
||||
"if delete_storage or os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Delete the created GCS bucket\n",
|
||||
" ! gsutil rm -r $BUCKET_URI\n",
|
||||
" # Delete the created BigQuery datasets\n",
|
||||
" ! bq rm -r -f $PROJECT_ID:$BQ_MY_DATASET"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -33,14 +33,20 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_dataflow.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -139,7 +145,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages to execute this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -150,18 +156,26 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install -U tensorflow==2.5 $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-transform==1.2 $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-io==0.18 $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-bigquery $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade apache-beam[gcp] $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -193,6 +207,32 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "84cd83853240"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, BigQuery, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,bigquery,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -258,7 +298,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -269,7 +309,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -296,6 +339,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "77c385f0db59"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "535223fa4b84"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -324,7 +428,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -335,8 +440,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -356,7 +462,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -376,7 +482,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -399,7 +505,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -474,7 +580,7 @@
|
||||
"id": "import_numpy"
|
||||
},
|
||||
"source": [
|
||||
"#### Import pandas\n",
|
||||
"#### Import numpy\n",
|
||||
"\n",
|
||||
"Import the numpy package into your Python environment."
|
||||
]
|
||||
@@ -540,9 +646,9 @@
|
||||
"id": "init_aip:mbsdk,region"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex SDK for Python\n",
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -553,7 +659,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION)"
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -674,6 +780,31 @@
|
||||
"dataframe[\"station_number\"] = pd.to_numeric(dataframe[\"station_number\"])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"source": [
|
||||
"### Create BQ dataset resource\n",
|
||||
"\n",
|
||||
"First, you create an empty dataset resource in your project."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BQ_MY_DATASET = 'samples'\n",
|
||||
"BQ_MY_TABLE = 'gsod'\n",
|
||||
"! bq --location=US mk -d \\\n",
|
||||
"$PROJECT_ID:$BQ_MY_DATASET"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -977,7 +1108,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"SCHEMA_LOCATION = BUCKET_NAME + \"/schema.txt\"\n",
|
||||
"SCHEMA_LOCATION = BUCKET_URI + \"/schema.txt\"\n",
|
||||
"\n",
|
||||
"# When running Apache Beam directly (file is directly accessed)\n",
|
||||
"tfdv.write_schema_text(output_path=SCHEMA_LOCATION, schema=schema)\n",
|
||||
@@ -1122,7 +1253,7 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"EXPORTED_DATA_PREFIX = os.path.join(BUCKET_NAME, \"exported_data\")\n",
|
||||
"EXPORTED_DATA_PREFIX = os.path.join(BUCKET_URI, \"exported_data\")\n",
|
||||
"\n",
|
||||
"QUERY_STRING = \"SELECT {},{} FROM {} LIMIT 500\".format(\n",
|
||||
" \"CAST(station_number as STRING) AS station_number,year,month,day\",\n",
|
||||
@@ -1135,7 +1266,7 @@
|
||||
" \"runner\": RUNNER,\n",
|
||||
" \"raw_data_query\": QUERY_STRING,\n",
|
||||
" \"exported_data_prefix\": EXPORTED_DATA_PREFIX,\n",
|
||||
" \"temp_location\": os.path.join(BUCKET_NAME, \"temp\"),\n",
|
||||
" \"temp_location\": os.path.join(BUCKET_URI, \"temp\"),\n",
|
||||
" \"project\": PROJECT_ID,\n",
|
||||
" \"region\": REGION,\n",
|
||||
" \"setup_file\": \"./setup.py\",\n",
|
||||
@@ -1160,17 +1291,7 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1181,60 +1302,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"delete_storage = True\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_storage or os.getenv(\"IS_TESTING\"):\n",
|
||||
" if \"BUCKET_URI\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,16 +33,22 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
"<img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
@@ -67,16 +73,16 @@
|
||||
"source": [
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this tutorial, you learn how to use `Vertex Dataset` for training with `Vertex AI`.\n",
|
||||
"In this tutorial, you learn how to use `Vertex AI Dataset` for training with `Vertex AI`.\n",
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex Datasets`\n",
|
||||
"- `Vertex AI Datasets`\n",
|
||||
"- `BigQuery Datasets`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Create a Vertex `Dataset` resource for:\n",
|
||||
"- Create a Vertex AI `Dataset` resource for:\n",
|
||||
" - image data\n",
|
||||
" - text data\n",
|
||||
" - video data\n",
|
||||
@@ -100,7 +106,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with Vertex Datasets:\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with Vertex AI Datasets:\n",
|
||||
"\n",
|
||||
"- Use CSV index file format for image data\n",
|
||||
"- Use CSV index file format for text data:\n",
|
||||
@@ -116,7 +122,7 @@
|
||||
"\n",
|
||||
"- Use `filter` and `order_by` parameters in the `list()` methods to find the latest versions of datasets.\n",
|
||||
"\n",
|
||||
"- When custom training with a `Vertex Dataset`:\n",
|
||||
"- When custom training with a `Vertex AI Dataset`:\n",
|
||||
" - tabular data :\n",
|
||||
" - Use the CSV index file or BigQuery table reference.\n",
|
||||
" - Create a tf.data.Dataset generator from the CSV index file/BigQuery table.\n",
|
||||
@@ -130,7 +136,17 @@
|
||||
" - Create a tf.data.Dataset generator from the CSV index file.\n",
|
||||
" - If text strings are in text files:\n",
|
||||
" - Using the JSON index file, convert the text files and labels to TFRecords.\n",
|
||||
" - Create a tf.data.Dataset from the TFRecords."
|
||||
" - Create a tf.data.Dataset from the TFRecords.\n",
|
||||
"\n",
|
||||
" \n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"- BigQuery\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -141,7 +157,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the packages required for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -152,18 +168,27 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install -U tensorflow $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-data-validation $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-transform $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-io $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-bigquery $USER_FLAG -q\n",
|
||||
"! pip3 install -U tensorflow-io==0.18 $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade db-dtypes $USER_FLAG -q! pip3 install --upgrade future $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -195,6 +220,30 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cb082379ed5b"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com). \n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -214,7 +263,24 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Get your Google Cloud project ID from gcloud\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "37c0a68ff20d"
|
||||
},
|
||||
"source": [
|
||||
"Otherwise, set your project ID here."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -225,18 +291,15 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID:\", PROJECT_ID)"
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
|
||||
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_gcloud_project_id"
|
||||
"id": "c021ca495967"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -260,7 +323,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -271,7 +334,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -298,6 +364,66 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "927085b84a07"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"**Click Create service account**.\n",
|
||||
"\n",
|
||||
"In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "89788a802687"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = \"google.colab\" in sys.modules\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -326,7 +452,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -337,8 +463,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -358,7 +484,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -378,7 +504,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -390,7 +516,11 @@
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
"### Import libraries and define constants\n",
|
||||
"\n",
|
||||
"Import the BigQuery package, TensorFlow Data Validation (TFDV) package and TensorFlow Data Validation package into your Python environment. \n",
|
||||
"\n",
|
||||
"Import TensorFlow Transform (TFT) package and pandas into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -401,106 +531,22 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"source": [
|
||||
"#### Import BigQuery\n",
|
||||
"\n",
|
||||
"Import the BigQuery package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip\n",
|
||||
"import pandas as pd\n",
|
||||
"import tensorflow_data_validation as tfdv\n",
|
||||
"import tensorflow_transform as tft\n",
|
||||
"from google.cloud import bigquery"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_tfdv"
|
||||
},
|
||||
"source": [
|
||||
"#### Import TensorFlow Data Validation\n",
|
||||
"\n",
|
||||
"Import the TensorFlow Data Validation (TFDV) package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_tfdv"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import tensorflow_data_validation as tfdv"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_tft"
|
||||
},
|
||||
"source": [
|
||||
"#### Import TensorFlow Transform\n",
|
||||
"\n",
|
||||
"Import the TensorFlow Transform (TFT) package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_tft"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import tensorflow_transform as tft"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_pandas"
|
||||
},
|
||||
"source": [
|
||||
"#### Import pandas\n",
|
||||
"\n",
|
||||
"Import the pandas package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_pandas"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pandas as pd"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,region"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex SDK for Python\n",
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -511,7 +557,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION)"
|
||||
"aip.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -542,9 +588,9 @@
|
||||
"id": "dataset_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Vertex Datasets\n",
|
||||
"## Vertex AI Datasets\n",
|
||||
"\n",
|
||||
"Vertex `Datasets` are the means for managing your datasets within Vertex AI services. Vertex Datasets are also referred to as `Dataset` resources. There are four types of `Dataset` resources, specific to the data type:\n",
|
||||
"Vertex AI `Datasets` are the means for managing your datasets within Vertex AI services. Vertex AI Datasets are also referred to as `Dataset` resources. There are four types of `Dataset` resources, specific to the data type:\n",
|
||||
"\n",
|
||||
"- `ImageDataset`: image data\n",
|
||||
"- `TabularDataset`: tabular (structured) data\n",
|
||||
@@ -552,7 +598,7 @@
|
||||
"- `VideoDataset`: video data\n",
|
||||
"- `TimeSeriesDataset`: forecasting data\n",
|
||||
"\n",
|
||||
"A Vertex `Dataset` provides the following capabilities:\n",
|
||||
"A Vertex AI `Dataset` provides the following capabilities:\n",
|
||||
"\n",
|
||||
"- A unique internal identifier for automatic (programatic) processes.\n",
|
||||
"- A user specificed (display name) identifier for interactive processes.\n",
|
||||
@@ -565,26 +611,13 @@
|
||||
"Learn more about [All dataset documentation](https://cloud.google.com/vertex-ai/docs/datasets/datasets)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:flowers,csv,icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = (\n",
|
||||
" \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_dataset:image,icn"
|
||||
},
|
||||
"source": [
|
||||
"### Create the Dataset\n",
|
||||
"### Create an Image Dataset\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `ImageDataset` class, which takes the following parameters:\n",
|
||||
"\n",
|
||||
@@ -599,6 +632,19 @@
|
||||
"Learn more about [ImageDataset](https://cloud.google.com/vertex-ai/docs/datasets/prepare-image)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:flowers,csv,icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = (\n",
|
||||
" \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -616,24 +662,13 @@
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:hmdb,csv,vcn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_dataset:video,vcn"
|
||||
},
|
||||
"source": [
|
||||
"### Create the Dataset\n",
|
||||
"### Create a Video Dataset\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `VideoDataset` class, which takes the following parameters:\n",
|
||||
"\n",
|
||||
@@ -647,6 +682,17 @@
|
||||
"Learn more about [VideoDataset](https://cloud.google.com/vertex-ai/docs/datasets/prepare-video)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:hmdb,csv,vcn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -664,24 +710,13 @@
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:happydb,csv,tcn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_dataset:text,tcn"
|
||||
},
|
||||
"source": [
|
||||
"### Create the Dataset\n",
|
||||
"### Create a Text Dataset\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `TextDataset` class, which takes the following parameters:\n",
|
||||
"\n",
|
||||
@@ -696,6 +731,17 @@
|
||||
"Learn more about [TextDataset](https://cloud.google.com/vertex-ai/docs/datasets/prepare-text)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:happydb,csv,tcn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -713,6 +759,24 @@
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_dataset:tabular,bq,lrg,v2"
|
||||
},
|
||||
"source": [
|
||||
"### Create a Tabular Dataset\n",
|
||||
"\n",
|
||||
"#### CSV input data\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `TabularDataset` class for CSV input data, which takes the following parameters:\n",
|
||||
"\n",
|
||||
"- `display_name`: The human readable name for the `Dataset` resource.\n",
|
||||
"- `gcs_source`: A list of one or more dataset index files to import the data items into the `Dataset` resource.\n",
|
||||
"\n",
|
||||
"Learn more about [TabularDataset from CSV files](https://cloud.google.com/vertex-ai/docs/datasets/create-dataset-api#aiplatform_create_dataset_tabular_gcs_sample-python)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -721,27 +785,50 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"bq://bigquery-public-data.samples.gsod\"\n",
|
||||
"BQ_TABLE = \"bigquery-public-data.samples.gsod\""
|
||||
"IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_dataset:tabular,bq,lrg,v2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
" display_name=\"example\" + \"_\" + TIMESTAMP, gcs_source=[IMPORT_FILE]\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_dataset:tabular,bq,lrg,v2"
|
||||
"id": "854dd1e0195c"
|
||||
},
|
||||
"source": [
|
||||
"### Create the Dataset\n",
|
||||
"#### BigQuery input data\n",
|
||||
"\n",
|
||||
"#### CSV input data\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `TabularDataset` class, which takes the following parameters:\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `TabularDataset` class for BigQuery table input, which takes the following parameters:\n",
|
||||
"\n",
|
||||
"- `display_name`: The human readable name for the `Dataset` resource.\n",
|
||||
"- `gcs_source`: A list of one or more dataset index files to import the data items into the `Dataset` resource.\n",
|
||||
"- `labels`: User defined metadata. In this example, you store the location of the Cloud Storage bucket containing the user defined data.\n",
|
||||
"- `bq_source`: A list of one or more BigQuery tables to import the data items into the `Dataset` resource.\n",
|
||||
"\n",
|
||||
"Learn more about [TabularDataset from CSV files](https://cloud.google.com/vertex-ai/docs/datasets/create-dataset-api#aiplatform_create_dataset_tabular_gcs_sample-python)"
|
||||
"Learn more about [TabularDataset from BigQuery table](https://cloud.google.com/vertex-ai/docs/datasets/create-dataset-api#aiplatform_create_dataset_tabular_bigquery_sample-pythonn)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "86343c146300"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"bq://bigquery-public-data.samples.gsod\"\n",
|
||||
"BQ_TABLE = \"bigquery-public-data.samples.gsod\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -759,15 +846,63 @@
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "82e9fe20ce71"
|
||||
},
|
||||
"source": [
|
||||
"#### Dataframe input data\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create_from_dataframe` method for the `TabularDataset` class for pandas dataframe input, which takes the following parameters:\n",
|
||||
"\n",
|
||||
"- `display_name`: The human readable name for the `Dataset` resource.\n",
|
||||
"- `df_source`: The pandas dataframe to import the data items into the `Dataset` resource.\n",
|
||||
"- `staging_path`: The BigQuery table to store the imported data."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:covid,csv,forecast"
|
||||
"id": "3805f945ffdd"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"gs://cloud-samples-data/ai-platform/covid/bigquery-public-covid-nyt-us-counties-train.csv\""
|
||||
"# Download the table.\n",
|
||||
"table = bigquery.TableReference.from_string(BQ_TABLE)\n",
|
||||
"\n",
|
||||
"rows = bqclient.list_rows(\n",
|
||||
" table,\n",
|
||||
" max_results=10000,\n",
|
||||
" selected_fields=[\n",
|
||||
" bigquery.SchemaField(\"station_number\", \"STRING\"),\n",
|
||||
" bigquery.SchemaField(\"year\", \"INTEGER\"),\n",
|
||||
" bigquery.SchemaField(\"month\", \"INTEGER\"),\n",
|
||||
" bigquery.SchemaField(\"day\", \"INTEGER\"),\n",
|
||||
" bigquery.SchemaField(\"mean_temp\", \"FLOAT\"),\n",
|
||||
" ],\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"dataframe = rows.to_dataframe()\n",
|
||||
"print(dataframe.head())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_dataset:tabular,bq,lrg,v2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TabularDataset.create_from_dataframe(\n",
|
||||
" display_name=\"example\" + \"_\" + TIMESTAMP,\n",
|
||||
" df_source=dataframe,\n",
|
||||
" staging_path=f\"bq://{PROJECT_ID}.samples.gsod\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -776,7 +911,7 @@
|
||||
"id": "create_dataset:tabular,forecast,v2"
|
||||
},
|
||||
"source": [
|
||||
"### Create the Dataset\n",
|
||||
"### Create a Time Series Dataset\n",
|
||||
"\n",
|
||||
"Next, create the `Dataset` resource using the `create` method for the `TimeSeriesDataset` class, which takes the following parameters:\n",
|
||||
"\n",
|
||||
@@ -787,6 +922,17 @@
|
||||
"Learn more about [TimeSeriesDataset](https://cloud.google.com/vertex-ai/docs/datasets/prepare-tabular)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:covid,csv,forecast"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = \"gs://cloud-samples-data/ai-platform/covid/bigquery-public-covid-nyt-us-counties-train.csv\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -809,7 +955,7 @@
|
||||
"id": "dataset_intro:methods"
|
||||
},
|
||||
"source": [
|
||||
"## Vertex Dataset properties and methods\n",
|
||||
"## Vertex AI Dataset properties and methods\n",
|
||||
"\n",
|
||||
"The following are the `Dataset` methods:\n",
|
||||
"\n",
|
||||
@@ -898,7 +1044,7 @@
|
||||
"source": [
|
||||
"## TensorFlow Data Validation\n",
|
||||
"\n",
|
||||
"The TensorFlow Data Validation (TFDV) package is used in conjunction with Vertex and BigQuery datasets for:\n",
|
||||
"The TensorFlow Data Validation (TFDV) package is used in conjunction with Vertex AI and BigQuery datasets for:\n",
|
||||
"\n",
|
||||
"- Generating dataset statistics.\n",
|
||||
"- Generating data schema for data validation.\n",
|
||||
@@ -1147,9 +1293,9 @@
|
||||
"comps = BQ_TABLE.split(\".\")\n",
|
||||
"BQ_PROJECT_DATASET_TABLE = comps[0] + \":\" + comps[1] + \".\" + comps[2]\n",
|
||||
"\n",
|
||||
"! bq --location=us extract --destination_format CSV $BQ_PROJECT_DATASET_TABLE $BUCKET_NAME/mydata*.csv\n",
|
||||
"! bq --location=us extract --destination_format CSV $BQ_PROJECT_DATASET_TABLE $BUCKET_URI/mydata*.csv\n",
|
||||
"\n",
|
||||
"IMPORT_FILES = ! gsutil ls $BUCKET_NAME/mydata*.csv\n",
|
||||
"IMPORT_FILES = ! gsutil ls $BUCKET_URI/mydata*.csv\n",
|
||||
"\n",
|
||||
"print(IMPORT_FILES)\n",
|
||||
"\n",
|
||||
@@ -1207,6 +1353,38 @@
|
||||
"To create a dataframe from multiple CSV sources, you read each CSV file and concatenate the dataframes together."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bcd2e4e0703b"
|
||||
},
|
||||
"source": [
|
||||
"If you are running this notebook on Colab, run the following cell to install packages fsspec and gcsfs."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "927bd3f92268"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Workbench AI Notebook, then don't execute this code\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" ! pip3 install fsspec\n",
|
||||
" ! pip3 install gcsfs"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -1267,7 +1445,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPORTED_DIR = f\"{BUCKET_NAME}/exported\"\n",
|
||||
"EXPORTED_DIR = f\"{BUCKET_URI}/exported\"\n",
|
||||
"exported_files = dataset.export_data(output_dir=EXPORTED_DIR)\n",
|
||||
"\n",
|
||||
"! gsutil ls $EXPORTED_DIR"
|
||||
@@ -1496,7 +1674,7 @@
|
||||
" data = f.readlines()\n",
|
||||
"\n",
|
||||
"# The path to the TFRecord cached file.\n",
|
||||
"GCS_TFRECORD_URI = BUCKET_NAME + \"/flowers.tfrecord\"\n",
|
||||
"GCS_TFRECORD_URI = BUCKET_URI + \"/flowers.tfrecord\"\n",
|
||||
"\n",
|
||||
"# Create the TFRecord cached file\n",
|
||||
"with tf.io.TFRecordWriter(GCS_TFRECORD_URI) as writer:\n",
|
||||
@@ -1530,14 +1708,7 @@
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"- Bucket"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1548,60 +1719,16 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the dataset using the Vertex dataset object\n",
|
||||
"datasets = aip.TabularDataset.list(filter=f'display_name=\"example_{TIMESTAMP}\"')\n",
|
||||
"for dataset in datasets:\n",
|
||||
" dataset.delete()\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"# Delete the bucket\n",
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -0,0 +1,994 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "copyright"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
"# You may obtain a copy of the License at\n",
|
||||
"#\n",
|
||||
"# https://www.apache.org/licenses/LICENSE-2.0\n",
|
||||
"#\n",
|
||||
"# Unless required by applicable law or agreed to in writing, software\n",
|
||||
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
|
||||
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
|
||||
"# See the License for the specific language governing permissions and\n",
|
||||
"# limitations under the License."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "4JIDiHvGasba"
|
||||
},
|
||||
"source": [
|
||||
"This notebook was contributed by [Mohammad Al-Ansari](https://github.com/Mansari)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2xDiUNIZINWp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 1 : data management: create an unlabelled Vertex AI AutoML text entity extraction dataset from PDFs using Vision API\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_with_visionapi_and_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_with_visionapi_and_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_with_visionapi_and_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "H0alLPo_A-LK"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"This notebook will create an unlabelled `Vertex AI AutoML` text entity extraction dataset based on a collection of PDF files stored in a Cloud Storage bucket. \n",
|
||||
"\n",
|
||||
"The notebook can be modified to create different types of text datasets including sentiment analysis and classification."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "W4IBLTKOA5nl"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Patent PDF Samples with Extracted Structured Data](https://console.cloud.google.com/marketplace/product/global-patents/labeled-patents) from Google Public Data Sets. \n",
|
||||
"\n",
|
||||
"This dataset includes data extracted from over 300 patent documents issued in the US and EU. The dataset includes links to Cloud Storage blobs for the first page of each patent, in addition to a number of extracted entities. \n",
|
||||
"\n",
|
||||
"The data is published as a [public dataset](https://cloud.google.com/bigquery/public-data) on `BigQuery`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3f8c2f702ccd"
|
||||
},
|
||||
"source": [
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this tutorial, you learn to use `Vision API` to extract text from PDF files stored on a Cloud Storage bucket. You will then process the results and create an unlabelled `Vertex AI Dataset`, compatible with `AutoML`, for text entity extraction.\n",
|
||||
"\n",
|
||||
"You can then either use Google Cloud console to annotate / label the dataset, or create a labelling job as demonstrated in [this notebook](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_with_data_labeling.ipynb).\n",
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud services:\n",
|
||||
"\n",
|
||||
"- `Vision AI`\n",
|
||||
"- `Vertex AI AutoML`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"1. Using `Vision API` to perform Optical Character Recognition (OCR) to extract text from PDF files.\n",
|
||||
"2. Processing the results and saving them to text files.\n",
|
||||
"3. Generating a `Vertex AI Dataset` import file.\n",
|
||||
"4. Creating a new unlabelled text entity extraction `Vertex AI Dataset` resource in `Vertex AI`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "CgLDJ419LPJs"
|
||||
},
|
||||
"source": [
|
||||
"### Costs\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vision API\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing), [Vision API pricing](https://cloud.google.com/vision/pricing), [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "va2g7m9wLTjA"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your local development environment\n",
|
||||
"\n",
|
||||
"If you are using Colab or Vertex AI Workbench Notebooks, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
|
||||
"\n",
|
||||
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
|
||||
"\n",
|
||||
"- The Vision API SDK\n",
|
||||
"- The Vertex AI SDK\n",
|
||||
"- The Cloud Storage SDK\n",
|
||||
"- Git\n",
|
||||
"- Python 3\n",
|
||||
"- virtualenv\n",
|
||||
"- Jupyter notebook running in a virtual environment with Python 3\n",
|
||||
"\n",
|
||||
"The Cloud Storage guide to [Setting up a Python development environment](https://cloud.google.com/python/setup) and the [Jupyter installation guide](https://jupyter.org/install) provide detailed instructions for meeting these requirements. The following steps provide a condensed set of instructions:\n",
|
||||
"\n",
|
||||
"1. [Install and initialize the SDKs](https://cloud.google.com/sdk/docs/).\n",
|
||||
"\n",
|
||||
"2. [Install Python 3](https://cloud.google.com/python/setup#installing_python).\n",
|
||||
"\n",
|
||||
"3. [Install virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv) and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
|
||||
"\n",
|
||||
"4. To install Jupyter, run `pip3 install jupyter` on the command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"5. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"6. Open this notebook in the Jupyter Notebook Dashboard.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "X2tZAmugAe6h"
|
||||
},
|
||||
"source": [
|
||||
"## Installation\n",
|
||||
"\n",
|
||||
"Install the packages required for executing this notebook. You can ignore errors for the `pip` dependecy resolver as they do not impact this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "BQOsJ1hZAZu0"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-storage google-cloud-vision google-cloud-aiplatform $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "yzvvcmCuAon3"
|
||||
},
|
||||
"source": [
|
||||
"### Restart the kernel\n",
|
||||
"\n",
|
||||
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "6qEonzbuAoI_"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Automatically restart kernel after installs\n",
|
||||
" import IPython\n",
|
||||
"\n",
|
||||
" app = IPython.Application.instance()\n",
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "pGbbyN7rAuRM"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### GPU runtime\n",
|
||||
"\n",
|
||||
"*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n",
|
||||
"\n",
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
|
||||
"\n",
|
||||
"3. [Enable the following APIs: Vision API, Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=vision.googleapis.com,aiplatform.googleapis.com,compute_component,storage-component.googleapis.com)\n",
|
||||
"\n",
|
||||
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
|
||||
"\n",
|
||||
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "AE97adtnAzrr"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "nWlzLu5ELxWd"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID:\", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "GB5b27r0LxqE"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud config set project $PROJECT_ID"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "pMJdU1K5xG7D"
|
||||
},
|
||||
"source": [
|
||||
"### Regions\n",
|
||||
"\n",
|
||||
"#### Vision AI\n",
|
||||
"\n",
|
||||
"You can now specify continent-level data storage and Optical Character Regonition (OCR) processing by setting the `VISION_AI_REGION` variable. You can select one of the following options:\n",
|
||||
"\n",
|
||||
"* USA country only: `us`\n",
|
||||
"* The European Union: `eu`\n",
|
||||
"\n",
|
||||
"Learn more about [Vision AI regions for OCR](https://cloud.google.com/vision/docs/pdf#regionalization)\n",
|
||||
"\n",
|
||||
"#### Vertex AI\n",
|
||||
"\n",
|
||||
"You can also change the `VERTEX_AI_REGION` variable, which is used for operations throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
|
||||
"\n",
|
||||
"- Americas: `us-central1`\n",
|
||||
"- Europe: `europe-west4`\n",
|
||||
"- Asia Pacific: `asia-east1`\n",
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "5EhEAOK5xIKc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"VISION_AI_REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if VISION_AI_REGION == \"[your-region]\":\n",
|
||||
" VISION_AI_REGION = \"us\"\n",
|
||||
"\n",
|
||||
"VERTEX_AI_REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if VERTEX_AI_REGION == \"[your-region]\":\n",
|
||||
" VERTEX_AI_REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "xkgvWoXkxM1r"
|
||||
},
|
||||
"source": [
|
||||
"### Timestamp\n",
|
||||
"\n",
|
||||
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "gr0HTpQZxNy4"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "AA-ns5CcBA9U"
|
||||
},
|
||||
"source": [
|
||||
"### Vertex AI dataset import schema\n",
|
||||
"\n",
|
||||
"This constant tells Vertex AI the schema for importing the dataset. In this tutorial you are going to use the value for text extraction, but you can also change it to any of the values below for other use cases:\n",
|
||||
"\n",
|
||||
"- \n",
|
||||
"`aiplatform.schema.dataset.ioformat.text.single_label_classification`\n",
|
||||
"\n",
|
||||
"- \n",
|
||||
"`aiplatform.schema.dataset.ioformat.text.multi_label_classification`\n",
|
||||
"\n",
|
||||
"- \n",
|
||||
"`aiplatform.schema.dataset.ioformat.text.extraction`\n",
|
||||
"\n",
|
||||
"- \n",
|
||||
"`aiplatform.schema.dataset.ioformat.text.sentiment`\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "jnOb6Pp-4w5P"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"DATASET_IMPORT_SCHEMA = aiplatform.schema.dataset.ioformat.text.extraction"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "ekbg-G7UA-bK"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench**, your environment is already authenticated. Skip this step. If you receive errors still, you may have to grant the service account that is your Workbench notebook is running under access to the services listed below.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"**Click Create service account**.\n",
|
||||
"\n",
|
||||
"In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "lCRrULxKBAfa"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = \"google.colab\" in sys.modules\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "rHB6fbonMMbI"
|
||||
},
|
||||
"source": [
|
||||
"### Create a Cloud Storage bucket\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"When you initialize the Vertex AI SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions. This bucket will be also used to store the output of the Vision API SDK PDF-to-text conversion process.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ZSM5j0nfMOVK"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "i6H2iQX2MP-s"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "AOsnYE5cMQX4"
|
||||
},
|
||||
"source": [
|
||||
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "33RgSjhyMR6C"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $VERTEX_AI_REGION -p $PROJECT_ID $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "UpKfi0VfMTwe"
|
||||
},
|
||||
"source": [
|
||||
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "G9dMjMnkMVNt"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "k2qH7YCI0vnG"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "TB5-_2Xh01NH"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform, storage, vision"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "-v7gY_KABIn8"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vision API SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the `Vision AI` SDK for Python for your project and region."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "DRbf--kWBLpx"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"vision_client_options = {\n",
|
||||
" \"quota_project_id\": PROJECT_ID,\n",
|
||||
" \"api_endpoint\": f\"{VISION_AI_REGION}-vision.googleapis.com\",\n",
|
||||
"}\n",
|
||||
"vision_client = vision.ImageAnnotatorClient(client_options=vision_client_options)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "CA4nNVbBZ25d"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the `Vertex AI` SDK for Python for your project, region and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "awWpNW1vZ6uV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aiplatform.init(\n",
|
||||
" project=PROJECT_ID, location=VERTEX_AI_REGION, staging_bucket=BUCKET_URI\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "debBBljMDqkM"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Cloud Storage SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the `Cloud Storage` SDK for Python for your project."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ZtzmI9tpDr4e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"storage_client = storage.Client(project=PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "mvD0BxVXMtJe"
|
||||
},
|
||||
"source": [
|
||||
"## Tutorial\n",
|
||||
"\n",
|
||||
"Now you are ready to start creating an unlabelled `Vertex AI Dataset` text entity extraction dataset from PDF files."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "EurEFM3GBap9"
|
||||
},
|
||||
"source": [
|
||||
"### Convert PDF files to text using Vision API\n",
|
||||
"\n",
|
||||
"First, you make a `Vision API` request to OCR to text the PDFs from the Patent samples stored in the Cloud Storage bucket.\n",
|
||||
"\n",
|
||||
"*Note:* `Visions API` only allows batches of 100 document submissions at a time."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "uXVPOvjTBeK3"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ORIGIN_BUCKET_NAME = \"gcs-public-data--labeled-patents\"\n",
|
||||
"# You can add a path if needed\n",
|
||||
"ORIGIN_BUCKET_PATH = \"\"\n",
|
||||
"\n",
|
||||
"DESTINATION_BUCKET_NAME = BUCKET_NAME\n",
|
||||
"DESTINATION_BUCKET_PATH = \"ocr-output\"\n",
|
||||
"\n",
|
||||
"gcs_destination_uri = f\"gs://{DESTINATION_BUCKET_NAME}/{DESTINATION_BUCKET_PATH}\"\n",
|
||||
"\n",
|
||||
"# Specify the feature for the Vision API processor\n",
|
||||
"feature = vision.Feature(type_=vision.Feature.Type.DOCUMENT_TEXT_DETECTION)\n",
|
||||
"\n",
|
||||
"# Retrieve a list of all files in the bucket and path\n",
|
||||
"blobs = storage_client.list_blobs(\n",
|
||||
" ORIGIN_BUCKET_NAME, prefix=ORIGIN_BUCKET_PATH, delimiter=\"/\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Create a collection of requests. The SDK requires a separate request per each\n",
|
||||
"# file that we want to extract text from\n",
|
||||
"async_requests = []\n",
|
||||
"\n",
|
||||
"# Visions API only supports processing up to 100 documents at a time\n",
|
||||
"# so we will process the first 100 elements only\n",
|
||||
"sliced_blob_list = list(blobs)[:100]\n",
|
||||
"\n",
|
||||
"# Loop through the source bucket and create a request for each file there\n",
|
||||
"for blob in sliced_blob_list:\n",
|
||||
" # Build input_config\n",
|
||||
" # Ensure we are only processing PDF files\n",
|
||||
" if blob.name.endswith(\".pdf\"):\n",
|
||||
" gcs_source = vision.GcsSource(uri=f\"gs://{ORIGIN_BUCKET_NAME}/{blob.name}\")\n",
|
||||
" input_config = vision.InputConfig(\n",
|
||||
" gcs_source=gcs_source, mime_type=\"application/pdf\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Build output config\n",
|
||||
" # Get file name\n",
|
||||
" file_name = os.path.splitext(os.path.basename(blob.name))[0]\n",
|
||||
" gcs_destination = vision.GcsDestination(\n",
|
||||
" uri=f\"{gcs_destination_uri}/{file_name}-\"\n",
|
||||
" )\n",
|
||||
" output_config = vision.OutputConfig(gcs_destination=gcs_destination)\n",
|
||||
"\n",
|
||||
" # Build request object and add to the collection\n",
|
||||
" async_request = vision.AsyncAnnotateFileRequest(\n",
|
||||
" features=[feature], input_config=input_config, output_config=output_config\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" async_requests.append(async_request)\n",
|
||||
"\n",
|
||||
"print(f\"Created {len(async_requests)} requests\")\n",
|
||||
"\n",
|
||||
"# Submit the batch OCR job\n",
|
||||
"\n",
|
||||
"operation = vision_client.async_batch_annotate_files(requests=async_requests)\n",
|
||||
"print(\"Submitting the batch OCR job\")\n",
|
||||
"\n",
|
||||
"print(\"Waiting for the operation to finish... this will take a short while\")\n",
|
||||
"\n",
|
||||
"response = operation.result(timeout=420)\n",
|
||||
"\n",
|
||||
"print(\"Completed!\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "7b15473e1937"
|
||||
},
|
||||
"source": [
|
||||
"#### Quick peek at extracted annotated JSON files\n",
|
||||
"\n",
|
||||
"Next, you take a peek at the contents of one of the extracted JSON annotated files."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4366442c1373"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"json_files = ! gsutil ls {gcs_destination_uri}\n",
|
||||
"\n",
|
||||
"example = json_files[0]\n",
|
||||
"! gsutil cat {example} | head -n 1"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "QWmeHWPIHako"
|
||||
},
|
||||
"source": [
|
||||
"### Process results and build the import file\n",
|
||||
"\n",
|
||||
"The `Vision API` output is in JSON format, and contains detailed text extraction data. You only need the full text output, so you will processs the JSON results, extract the text output, and save it in new text files to be used later in the tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "WDLtiejKHug6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"\n",
|
||||
"print(\"Extracting text from Vision API output and saving it to text files\")\n",
|
||||
"\n",
|
||||
"ocr_blobs = storage_client.list_blobs(\n",
|
||||
" DESTINATION_BUCKET_NAME, prefix=DESTINATION_BUCKET_PATH\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"output_bucket = storage_client.bucket(DESTINATION_BUCKET_NAME)\n",
|
||||
"\n",
|
||||
"# begin building the import file content\n",
|
||||
"import_file_entries = []\n",
|
||||
"\n",
|
||||
"for ocr_blob in ocr_blobs:\n",
|
||||
" # Only process .json files, in case we previously processed files and had .txt files\n",
|
||||
" if ocr_blob.name.endswith(\".json\"):\n",
|
||||
" print(f\"Extracting text from {ocr_blob.name}\")\n",
|
||||
" # read each blob into a stream\n",
|
||||
" contents = ocr_blob.download_as_string()\n",
|
||||
" # load as JSON\n",
|
||||
" json_object = json.loads(contents)\n",
|
||||
" # extract text\n",
|
||||
" full_text = \"\"\n",
|
||||
" for response in json_object[\"responses\"]:\n",
|
||||
" if response[\"fullTextAnnotation\"]:\n",
|
||||
" full_text += response[\"fullTextAnnotation\"][\"text\"] + \"\\r\\n\"\n",
|
||||
"\n",
|
||||
" # save as a blob\n",
|
||||
" output_blob_name = f\"{ocr_blob.name}.txt\"\n",
|
||||
" import_file_blob = output_bucket.blob(output_blob_name)\n",
|
||||
" import_file_blob.upload_from_string(full_text)\n",
|
||||
"\n",
|
||||
" # create import file listing\n",
|
||||
" import_file_entry = {\n",
|
||||
" \"textGcsUri\": f\"gs://{DESTINATION_BUCKET_NAME}/{output_blob_name}\"\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" import_file_entries.append(import_file_entry)\n",
|
||||
"\n",
|
||||
"print(\"Extraction completed!\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "0a5aae0eab44"
|
||||
},
|
||||
"source": [
|
||||
"#### Quick peek at extracted text files\n",
|
||||
"\n",
|
||||
"Next, you take a peek at the contents of one of the extracted text files."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "76ce5f57b1ae"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"example = import_file_entries[0][\"textGcsUri\"]\n",
|
||||
"\n",
|
||||
"! gsutil cat {example}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "hqTLS_AmLWQP"
|
||||
},
|
||||
"source": [
|
||||
"### Generate and save import file to be used in `Vertex AI Dataset` resource\n",
|
||||
"\n",
|
||||
"You will now build the import file that will be used to create the `Vertex AI Dataset` resource."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "_xFvOdQ_LWne"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE_PATH = \"import_file\"\n",
|
||||
"\n",
|
||||
"# Convert import file entries to JSON Lines format\n",
|
||||
"import_file_content = \"\"\n",
|
||||
"for entry in import_file_entries:\n",
|
||||
" import_file_content += json.dumps(entry) + \"\\n\"\n",
|
||||
"\n",
|
||||
"print(f\"Created import file based on {len(import_file_entries)} annotations\")\n",
|
||||
"\n",
|
||||
"# Upload content to GCS to be used in our next step\n",
|
||||
"gcs_annotation_file_name = f\"{IMPORT_FILE_PATH}/import_file_{TIMESTAMP}.jsonl\"\n",
|
||||
"import_file_blob = output_bucket.blob(gcs_annotation_file_name)\n",
|
||||
"import_file_blob.upload_from_string(import_file_content)\n",
|
||||
"\n",
|
||||
"print(f\"Uploaded import file to {output_bucket.name}/{gcs_annotation_file_name}\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "6dVjFftOaKdw"
|
||||
},
|
||||
"source": [
|
||||
"### Create an unlabelled `Vertex AI Dataset` resource\n",
|
||||
"\n",
|
||||
"Next, you create the `Dataset` resource using the `create` method for the `TextDataset` class, which takes the following parameters:\n",
|
||||
"\n",
|
||||
"- `display_name`: The human readable name for the `Dataset` resource.\n",
|
||||
"- `gcs_source`: A list of one or more dataset index files to import the data items into the `Dataset` resource.\n",
|
||||
"- `import_schema_uri`: The data labeling schema for the data items.\n",
|
||||
"\n",
|
||||
"This operation may take ten to twenty minutes."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ciM9HLGCaOTJ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"print(\"Creating dataset ...\")\n",
|
||||
"\n",
|
||||
"dataset = aiplatform.TextDataset.create(\n",
|
||||
" display_name=\"Text Dataset \" + TIMESTAMP,\n",
|
||||
" gcs_source=[f\"gs://{output_bucket.name}/{gcs_annotation_file_name}\"],\n",
|
||||
" import_schema_uri=DATASET_IMPORT_SCHEMA,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"Completed!\")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2vagHf5T6Jd4"
|
||||
},
|
||||
"source": [
|
||||
"**Congratulations, your dataset is now ready for annotations!**\n",
|
||||
"\n",
|
||||
"You have two options:\n",
|
||||
"\n",
|
||||
"* Use Google Cloud Console to manually annotate the dataset in `Vertex AI`. Checkout [this link](https://cloud.google.com/vertex-ai/docs/datasets/label-using-console#entity-extraction) for more details on how to do so.\n",
|
||||
"* Create a labelling job to request data labelling. Check out [this link](https://cloud.google.com/vertex-ai/docs/datasets/data-labeling-job) and [this notebook](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_with_data_labeling.ipynb) for more details and examples.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cleanup:migration,new"
|
||||
},
|
||||
"source": [
|
||||
"# Cleaning up\n",
|
||||
"\n",
|
||||
"To clean up all GCP resources used in this project, you can [delete the GCP\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "aoJ18d8Y_jAy"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n",
|
||||
"dataset.delete()\n",
|
||||
"\n",
|
||||
"# Delete the bucket created\n",
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"accelerator": "GPU",
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "get_started_with_visionapi_and_vertex_datasets.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -33,14 +33,20 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/mlops_data_management.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/mlops_data_management.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/mlops_data_management.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/mlops_data_management.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage1/mlops_data_management.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -133,13 +139,27 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"ONCE_ONLY = True\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -177,6 +197,32 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "84cd83853240"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, BigQuery, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,bigquery,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -242,7 +288,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -253,7 +299,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -280,6 +329,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "77c385f0db59"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "535223fa4b84"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -308,7 +418,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -319,8 +430,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -340,7 +452,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -360,7 +472,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -515,7 +627,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -756,7 +868,7 @@
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
" display_name=\"Chicago Taxi\" + \"_\" + TIMESTAMP,\n",
|
||||
" bq_source=[IMPORT_FILE],\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"label_column = \"tip_bin\"\n",
|
||||
@@ -948,9 +1060,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"STATISTICS_SCHEMA = BUCKET_NAME + \"/statistics.jsonl\"\n",
|
||||
"STATISTICS_SCHEMA = BUCKET_URI + \"/statistics.jsonl\"\n",
|
||||
"\n",
|
||||
"tfdv.write_stats_text(stats, BUCKET_NAME + \"/statistics.jsonl\")\n",
|
||||
"tfdv.write_stats_text(stats, BUCKET_URI + \"/statistics.jsonl\")\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(\n",
|
||||
" \"gs://\" + dataset.labels[\"user_metadata\"] + \"/metadata.jsonl\", \"r\"\n",
|
||||
@@ -963,7 +1075,7 @@
|
||||
") as f:\n",
|
||||
" json.dump(metadata, f)\n",
|
||||
"\n",
|
||||
"!gsutil cat $BUCKET_NAME/metadata.jsonl"
|
||||
"! gsutil cat $BUCKET_URI/metadata.jsonl"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1010,7 +1122,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"SCHEMA_LOCATION = BUCKET_NAME + \"/schema.txt\"\n",
|
||||
"SCHEMA_LOCATION = BUCKET_URI + \"/schema.txt\"\n",
|
||||
"\n",
|
||||
"# When running Apache Beam directly (file is directly accessed)\n",
|
||||
"tfdv.write_schema_text(output_path=SCHEMA_LOCATION, schema=schema)\n",
|
||||
@@ -1048,7 +1160,7 @@
|
||||
") as f:\n",
|
||||
" json.dump(metadata, f)\n",
|
||||
"\n",
|
||||
"!gsutil cat $BUCKET_NAME/metadata.jsonl"
|
||||
"! gsutil cat $BUCKET_URI/metadata.jsonl"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1371,10 +1483,10 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"EXPORTED_JSONL_PREFIX = os.path.join(BUCKET_NAME, \"exported_data/jsonl\")\n",
|
||||
"EXPORTED_TFREC_PREFIX = os.path.join(BUCKET_NAME, \"exported_data/tfrec\")\n",
|
||||
"TRANSFORMED_DATA_PREFIX = os.path.join(BUCKET_NAME, \"transformed_data\")\n",
|
||||
"TRANSFORM_ARTIFACTS_DIR = os.path.join(BUCKET_NAME, \"transformed_artifacts\")\n",
|
||||
"EXPORTED_JSONL_PREFIX = os.path.join(BUCKET_URI, \"exported_data/jsonl\")\n",
|
||||
"EXPORTED_TFREC_PREFIX = os.path.join(BUCKET_URI, \"exported_data/tfrec\")\n",
|
||||
"TRANSFORMED_DATA_PREFIX = os.path.join(BUCKET_URI, \"transformed_data\")\n",
|
||||
"TRANSFORM_ARTIFACTS_DIR = os.path.join(BUCKET_URI, \"transformed_artifacts\")\n",
|
||||
"\n",
|
||||
"QUERY_STRING = \"SELECT * FROM {} LIMIT 300000\".format(BQ_TABLE)\n",
|
||||
"JOB_NAME = \"chicago\" + TIMESTAMP\n",
|
||||
@@ -1387,7 +1499,7 @@
|
||||
" \"transform_artifact_dir\": TRANSFORM_ARTIFACTS_DIR,\n",
|
||||
" \"exported_jsonl_prefix\": EXPORTED_JSONL_PREFIX,\n",
|
||||
" \"exported_tfrec_prefix\": EXPORTED_TFREC_PREFIX,\n",
|
||||
" \"temp_location\": os.path.join(BUCKET_NAME, \"temp\"),\n",
|
||||
" \"temp_location\": os.path.join(BUCKET_URI, \"temp\"),\n",
|
||||
" \"project\": PROJECT_ID,\n",
|
||||
" \"region\": REGION,\n",
|
||||
" \"setup_file\": \"./setup.py\",\n",
|
||||
@@ -1458,7 +1570,7 @@
|
||||
") as f:\n",
|
||||
" json.dump(metadata, f)\n",
|
||||
"\n",
|
||||
"!gsutil cat $BUCKET_NAME/metadata.jsonl"
|
||||
"! gsutil cat $BUCKET_URI/metadata.jsonl"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1472,17 +1584,9 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial.\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"*Note:* stage2/mlops_experimentation is dependent on the resources created by this stage1 notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1503,8 +1607,8 @@
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" if \"BUCKET_URI\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
|
Before Width: | Height: | Size: 39 KiB |
|
After Width: | Height: | Size: 76 KiB |
@@ -25,7 +25,11 @@ The second stage in MLOps is experimenting in developing one or more baseline mo
|
||||
- Use the What-if-Tool (WIT) to explore how the trained model would make predictions in different scenarios.
|
||||
|
||||
|
||||
<img src='stage2.png'>
|
||||
<img src='stage2v3.png'>
|
||||
<br/>
|
||||
<br/>
|
||||
<br/>
|
||||
<img src='stage2.2v1.png'>
|
||||
|
||||
## Notebooks
|
||||
|
||||
@@ -33,18 +37,250 @@ The second stage in MLOps is experimenting in developing one or more baseline mo
|
||||
|
||||
[Get Started with Vertex Experiments and Vertex ML Metadata](get_started_vertex_experiments.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Use Python logging to log training configuration/results locally.
|
||||
- Use Google Cloud Logging to log training configuration/results in cloud storage.
|
||||
- Create a Vertex AI `Experiment` resource.
|
||||
- Instantiate an experiment run.
|
||||
- Log parameters for the run.
|
||||
- Log metrics for the run.
|
||||
- Display the logged experiment run.
|
||||
```
|
||||
|
||||
[Get Started with Vertex TensorBoard](get_started_vertex_tensorboard.ipynb)
|
||||
|
||||
[Get Started with Custom Training Packages](get_started_vertex_training.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a TensorBoard callback when training a model.
|
||||
- Using Tensorboard with locally trained model.
|
||||
- Using Vertex AI TensorBoard with Vertex AI Training.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Tensorflow)](get_started_vertex_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a single Python script.
|
||||
- Training using a Python package.
|
||||
- Training using a custom training image.
|
||||
- Laying out a training package.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Scikit-Learn)](get_started_vertex_training_sklearn.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (XGBoost)](get_started_vertex_training_xgboost.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Pytorch)](get_started_vertex_training_pytorch.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Single node training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (R)](get_started_vertex_training_r.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Locally train an R model in a notebook using %%R magic commands
|
||||
- Create a deployment image with trained R model and serving functions.
|
||||
- Test the deployment image locally.
|
||||
- Create a `Vertex AI Model` resource for the deployment image with embedded R model.
|
||||
- Deploy the deployment image with embedded R model to a `Vertex AI Endpoint` resource.
|
||||
- Test the deployment image with embedded R model.
|
||||
- Create a R-to-Python training package.
|
||||
- Create a training image for training the model.
|
||||
- Train a R model using `Vertex AI Trainingh` service with the R-to-Python training package.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (R) and Deployment in R environment](get_started_vertex_training_r_using_r_kernel.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a custom R training script
|
||||
- Create a custom R serving script
|
||||
- Create a custom R deployment (serving) container.
|
||||
- Train the model using `Vertex AI` custom training.
|
||||
- Create an `Endpoint` resource.
|
||||
- Deploy the `Model` resource (trained R model) to the `Endpoint` resource.
|
||||
- Make an online prediction.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (LightGBM)](get_started_vertex_training_lightgbm.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Construct a FastAPI prediction server.
|
||||
- Construct a Dockerfile deployment image.
|
||||
- Test the deployment image locally.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Distributed Training](get_started_vertex_distributed_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- `MirroredStrategy`: Train on a single VM with multiple GPUs.
|
||||
- `MultiWorkerMirroredStrategy`: Train on multiple VMs with automatic setup of replicas.
|
||||
- `MultiWorkerMirroredStrategy`: Train on multiple VMs with fine grain control of replicas.
|
||||
- `ReductionServer`: Train on multiple VMS and sync updates across VMS with `Vertex AI Reduction Server`.
|
||||
- `TPUTraining`: Train with multiple Cloud TPUs.
|
||||
```
|
||||
|
||||
[Get Started with Vizier Hyperparameter Tuning](get_started_vertex_vizier.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Hyperparameter tuning with Random algorithm.
|
||||
- Hyperparameter tuning with Vizier (Bayesian) algorithm.
|
||||
```
|
||||
|
||||
[Get Started with AutoML Training](get_started_automl_training.ipynb)
|
||||
|
||||
[Get Started with BQML Training](get_started_bqml_training.ipyn)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Train an image model.
|
||||
- Export the image model as an edge model.
|
||||
- Train a tabular model.
|
||||
- Export the tabular model as a cloud model.
|
||||
- Train a text model.
|
||||
```
|
||||
|
||||
[Get Started with BQML Training](get_started_bqml_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a local BigQuery table in your project
|
||||
- Train a BQML model
|
||||
- Evaluate the BQML model
|
||||
- Export the BQML model as a cloud model
|
||||
- Upload the exported model as a `Vertex AI Model` resource
|
||||
- Hyperparameter tune a BQML model with `Vertex AI Vizier`
|
||||
- Automatically register a BQML model to `Vertex AI Model Registry`
|
||||
```
|
||||
|
||||
[Get Started with Vertex Feature Store](get_started_vertex_feature_store.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a Vertex AI `Featurestore` resource.
|
||||
- Creating `EntityType` resources for the `Featurestore` resource.
|
||||
- Creating `Feature` resources for each `EntityType` resource.
|
||||
- Import feature values (entity data items) into `Featurestore` resource from Cloud Storage.
|
||||
- Import feature values (entity data items) into `Featurestore` resource from pandas DataFrame.
|
||||
- Perform online serving from a `Featurestore` resource.
|
||||
- Perform batch serving from a `Featurestore` resource.
|
||||
```
|
||||
|
||||
[Get Started with Google CMEK Training](get_started_with_cmek_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a customer managed encryption key.
|
||||
- Creating an image dataset with CMEK encryption.
|
||||
- Train an AutoML model with CMEK encryption.
|
||||
```
|
||||
|
||||
[Get Started with TensorFlow Hub models](get_started_with_tfhub_models.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Download a TensorFlow Hub prebuilt model.
|
||||
- Add the task component as a classifier for the CIFAR-10 dataset.
|
||||
- Fine tune locally the model with transfer learning training.
|
||||
- Construct a custom training script:
|
||||
- Get training data from TensorFlow Datasets
|
||||
- Get model architecture from TensorFlow Hub
|
||||
- Train then model
|
||||
- Save model artifacts and upload as Vertex AI Model resource.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI TabNet builtin algorithm](get_started_with_tabnet.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Get the training data.
|
||||
- Configure training parameters for the Vertex AI TabNet container.
|
||||
- Train the model using Vertex AI Training using CSV data.
|
||||
- Upload the model as a Vertex AI Model resource.
|
||||
- Deploy the Vertex AI Model resource to a Vertex AI Endpoint resource.
|
||||
- Make a prediction with the deployed model.
|
||||
- Hyperparameter tuning the Vertex AI TabNet model.
|
||||
- Train the model using Vertex AI Training using BigQuery table.
|
||||
```
|
||||
|
||||
[Get Started with Vision API and AutoML](get_started_with_visionapi_and_automl.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Preprocess training files using `Vision AI` APIs to extract the text from PDF files.
|
||||
- Create a custom import file that includes annotation data based on the sample `BigQuery` dataset.
|
||||
- Create a `Vertex AI Dataset` resource.
|
||||
- Train the model.
|
||||
- View the model evaluation.
|
||||
- Deploy the `Vertex AI Model` resource to a serving `Endpoint` resource.
|
||||
- Make a prediction.
|
||||
- Undeploy the `Model`.
|
||||
```
|
||||
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 2: Experimentation](mlops_experimentation.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Review the `Dataset` resource created during stage 1.
|
||||
- Train an AutoML tabular binary classifier model in the background.
|
||||
- Build the experimental model architecture.
|
||||
- Construct a custom training package for the `Dataset` resource.
|
||||
- Test the custom training package locally.
|
||||
- Test the custom training package in the cloud with Vertex AI Training.
|
||||
- Hyperparameter tune the model training with Vertex AI Vizier.
|
||||
- Train the custom model with Vertex AI Training.
|
||||
- Add a serving function for online/batch prediction to the custom model.
|
||||
- Test the custom model with the serving function.
|
||||
- Evaluate the custom model using Vertex AI Batch Prediction
|
||||
- Wait for the AutoML training job to complete.
|
||||
- Evaluate the AutoML model using Vertex AI Batch Prediction with the same evaluation slices as the custom model.
|
||||
- Set the evaluation results of the AutoML model as the baseline.
|
||||
- If the evaluation of the custom model is below baseline, continue to experiment with the custom model.
|
||||
- If the evaluation of the custom model is above baseline, save the model as the first best model.
|
||||
```
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,14 +33,20 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_automl_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -65,9 +71,11 @@
|
||||
"id": "dataset:flowers,icn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"### Datasets\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip."
|
||||
"#### Image\n",
|
||||
"\n",
|
||||
"The image dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower in a given image from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -76,9 +84,9 @@
|
||||
"id": "dataset:gsod,lrg"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"#### Tabular\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
"The tabular dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -87,9 +95,20 @@
|
||||
"id": "dataset:happydb,tcn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"#### Text\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket."
|
||||
"The text dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "98eb93ec6faa"
|
||||
},
|
||||
"source": [
|
||||
"#### Video\n",
|
||||
"\n",
|
||||
"The video dataset used for this tutorial is the golf swing recognition portion of the [Human Motion dataset](https://todo) from [MIT](http://cbcl.mit.edu/publications/ps/Kuehne_etal_iccv11.pdf). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model will predict the start frame where a golf swing begins."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -108,11 +127,12 @@
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Train an image model.\n",
|
||||
"- Export the image model as an edge model.\n",
|
||||
"- Train a tabular model.\n",
|
||||
"- Export the tabular model as a cloud model.\n",
|
||||
"- Train a text model."
|
||||
"- Train an image model\n",
|
||||
"- Export the image model as an edge model\n",
|
||||
"- Train a tabular model\n",
|
||||
"- Export the tabular model as a cloud model\n",
|
||||
"- Train a text model\n",
|
||||
"- Train a video model"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -125,9 +145,24 @@
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are best practices for when to use AutoML:\n",
|
||||
"\n",
|
||||
"**You have a limited amount of training data**\n",
|
||||
"* **You have a limited amount of training data**\n",
|
||||
"\n",
|
||||
"**You want to establish a baseline metric before experimenting with a custom model**"
|
||||
"* **You want to establish a baseline metric before experimenting with a custom model**"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fb3451ce8e47"
|
||||
},
|
||||
"source": [
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -138,7 +173,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the packages required for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -149,19 +184,23 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"# Install the packages\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-storage $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -199,6 +238,23 @@
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n",
|
||||
"\n",
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
@@ -258,7 +314,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -269,7 +325,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -296,6 +355,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3ffa6b6c7cdb"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2b72272258fc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -319,7 +439,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -330,8 +451,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -351,7 +473,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -371,7 +493,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -394,7 +516,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -416,7 +538,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -447,7 +569,7 @@
|
||||
"source": [
|
||||
"## AutoML image models\n",
|
||||
"\n",
|
||||
"AutoML can train the following types of models:\n",
|
||||
"AutoML can train the following types of image models:\n",
|
||||
"\n",
|
||||
"- classification\n",
|
||||
"- objection detection\n",
|
||||
@@ -503,10 +625,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" FILE = IMPORT_FILES[0]\n",
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
@@ -544,10 +663,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.ImageDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.ImageDataset.create(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
@@ -592,8 +711,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLImageTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
"dag = aiplatform.AutoMLImageTrainingJob(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" multi_label=False,\n",
|
||||
" model_type=\"MOBILE_TF_LOW_LATENCY_1\",\n",
|
||||
@@ -611,14 +730,14 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
"- `training_fraction_split`: The percentage of the dataset to use for training.\n",
|
||||
"- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n",
|
||||
"- `validation_fraction_split`: The percentage of the dataset to use for validation.\n",
|
||||
"- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n",
|
||||
"- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of milli node-hours (1000 = node-hour).\n",
|
||||
"- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n",
|
||||
"\n",
|
||||
"The `run` method when completed returns the `Model` resource.\n",
|
||||
@@ -636,7 +755,7 @@
|
||||
"source": [
|
||||
"model = dag.run(\n",
|
||||
" dataset=dataset,\n",
|
||||
" model_display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" model_display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.8,\n",
|
||||
" validation_fraction_split=0.1,\n",
|
||||
" test_fraction_split=0.1,\n",
|
||||
@@ -652,9 +771,8 @@
|
||||
},
|
||||
"source": [
|
||||
"## Review model evaluation scores\n",
|
||||
"After your model has finished training, you can review the evaluation scores for it.\n",
|
||||
"\n",
|
||||
"First, you need to get a reference to the new model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project."
|
||||
"After your model training has finished, you can review the evaluation scores for it using the `list_model_evaluations()` method. This method will return an iterator for each evaluation slice."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -665,18 +783,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"model_evaluations = model.list_model_evaluations()\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
")\n",
|
||||
"model_evaluation = list(model_evaluations)[0]\n",
|
||||
"print(model_evaluation)"
|
||||
"for model_evaluation in model_evaluations:\n",
|
||||
" print(model_evaluation.to_dict())"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -720,7 +830,7 @@
|
||||
"source": [
|
||||
"### Get test item\n",
|
||||
"\n",
|
||||
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction."
|
||||
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model. You are just looking at how to make a prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -752,7 +862,7 @@
|
||||
"\n",
|
||||
"#### Request\n",
|
||||
"\n",
|
||||
"Since in this example your test item is in a Cloud Storage bucket, you open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, you encode the bytes into base64 -- which makes the content safe from modification while transmitting binary data over the network.\n",
|
||||
"Since your test item is in a public Cloud Storage bucket in this example, you copy it to your bucket and read the contents of the image using `Cloud Storage SDK`. To pass the test data to the prediction service, you encode the bytes into base64 which makes the content safe from modification while transmitting binary data over the network.\n",
|
||||
"\n",
|
||||
"The format of each instance is:\n",
|
||||
"\n",
|
||||
@@ -774,32 +884,66 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "predict_request:mbsdk,icn"
|
||||
"id": "1c1d53e89beb"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import base64\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"from google.cloud import storage\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"# Copy the test image to the Cloud storage bucket as \"test.jpg\"\n",
|
||||
"test_image_local = \"{}/test.jpg\".format(BUCKET_URI)\n",
|
||||
"! gsutil cp $test_item $test_image_local\n",
|
||||
"\n",
|
||||
"# Download the test image in bytes format\n",
|
||||
"storage_client = storage.Client(project=PROJECT_ID)\n",
|
||||
"bucket = storage_client.bucket(bucket_name=BUCKET_NAME)\n",
|
||||
"test_content = bucket.get_blob(\"test.jpg\").download_as_bytes()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"instances = [{\"content\": base64.b64encode(test_content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances)\n",
|
||||
"\n",
|
||||
"print(prediction)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3b1b67898533"
|
||||
},
|
||||
"source": [
|
||||
"#### Alternate method using [GFile](https://www.tensorflow.org/api_docs/python/tf/io/gfile/GFile)\n",
|
||||
"\n",
|
||||
"Alternatively, [GFile](https://www.tensorflow.org/api_docs/python/tf/io/gfile/GFile) method from tensorflow-io library can be used to read the data from Cloud storage directly. The following code snippet does the same :\n",
|
||||
"\n",
|
||||
"```\n",
|
||||
"import base64\n",
|
||||
"import tensorflow as tf\n",
|
||||
"\n",
|
||||
"# Read the test file using GFile\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances)\n",
|
||||
"\n",
|
||||
"print(prediction)\n",
|
||||
"```\n",
|
||||
"Nevertheless, `tf.io.gfile.GFile` supports multiple file system implementations, including local files, Google Cloud Storage (using a gs:// prefix), and HDFS (using an hdfs:// prefix)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -845,7 +989,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = model.export_model(\n",
|
||||
" artifact_destination=BUCKET_NAME, export_format_id=\"tflite\", sync=True\n",
|
||||
" artifact_destination=BUCKET_URI, export_format_id=\"tflite\", sync=True\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_package = response[\"artifactOutputUri\"]"
|
||||
@@ -986,10 +1130,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.TabularDataset.create(\n",
|
||||
" display_name=\"gsod_\" + TIMESTAMP,\n",
|
||||
" bq_source=[IMPORT_FILE],\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"label_column = \"mean_temp\"\n",
|
||||
@@ -1045,9 +1189,7 @@
|
||||
" - regression:\n",
|
||||
" - `minimize-rmse`\n",
|
||||
" - `minimize-mae`\n",
|
||||
" - `minimize-rmsle`\n",
|
||||
"\n",
|
||||
"The instantiated object is the DAG (directed acyclic graph) for the training pipeline."
|
||||
" - `minimize-rmsle`"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1058,8 +1200,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLTabularTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
"dag = aiplatform.AutoMLTabularTrainingJob(\n",
|
||||
" display_name=\"gsod_\" + TIMESTAMP,\n",
|
||||
" optimization_prediction_type=\"regression\",\n",
|
||||
" optimization_objective=\"minimize-rmse\",\n",
|
||||
" column_transformations=TRANSFORMATIONS,\n",
|
||||
@@ -1076,7 +1218,7 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
@@ -1102,7 +1244,7 @@
|
||||
"source": [
|
||||
"model = dag.run(\n",
|
||||
" dataset=dataset,\n",
|
||||
" model_display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" model_display_name=\"gsod_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.8,\n",
|
||||
" validation_fraction_split=0.1,\n",
|
||||
" test_fraction_split=0.1,\n",
|
||||
@@ -1119,9 +1261,8 @@
|
||||
},
|
||||
"source": [
|
||||
"## Review model evaluation scores\n",
|
||||
"After your model has finished training, you can review the evaluation scores for it.\n",
|
||||
"\n",
|
||||
"First, you need to get a reference to the new model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project."
|
||||
"After your model training has finished, you can review the evaluation scores for it using the `list_model_evaluations()` method. This method will return an iterator for each evaluation slice."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1132,18 +1273,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"model_evaluations = model.list_model_evaluations()\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
")\n",
|
||||
"model_evaluation = list(model_evaluations)[0]\n",
|
||||
"print(model_evaluation)"
|
||||
"for model_evaluation in model_evaluations:\n",
|
||||
" print(model_evaluation.to_dict())"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1176,7 +1309,7 @@
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -1217,7 +1350,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = model.export_model(\n",
|
||||
" artifact_destination=BUCKET_NAME, export_format_id=\"tf-saved-model\", sync=True\n",
|
||||
" artifact_destination=BUCKET_URI, export_format_id=\"tf-saved-model\", sync=True\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_package = response[\"artifactOutputUri\"]"
|
||||
@@ -1349,10 +1482,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" FILE = IMPORT_FILES[0]\n",
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
@@ -1390,10 +1520,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TextDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.TextDataset.create(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.text.single_label_classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.text.single_label_classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
@@ -1419,9 +1549,7 @@
|
||||
" - `sentiment`: A text sentiment analysis model.\n",
|
||||
" - `extraction`: A text entity extraction model.\n",
|
||||
"- `multi_label`: If a classification task, whether single (False) or multi-labeled (True).\n",
|
||||
"- `sentiment_max`: If a sentiment analysis task, the maximum sentiment value.\n",
|
||||
"\n",
|
||||
"The instantiated object is the DAG (directed acyclic graph) for the training pipeline."
|
||||
"- `sentiment_max`: If a sentiment analysis task, the maximum sentiment value.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1432,7 +1560,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLTextTrainingJob(\n",
|
||||
"dag = aiplatform.AutoMLTextTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" multi_label=False,\n",
|
||||
@@ -1449,7 +1577,7 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
@@ -1486,9 +1614,8 @@
|
||||
},
|
||||
"source": [
|
||||
"## Review model evaluation scores\n",
|
||||
"After your model has finished training, you can review the evaluation scores for it.\n",
|
||||
"\n",
|
||||
"First, you need to get a reference to the new model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project."
|
||||
"After your model training has finished, you can review the evaluation scores for it using the `list_model_evaluations()` method. This method will return an iterator for each evaluation slice."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1499,18 +1626,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"model_evaluations = model.list_model_evaluations()\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
")\n",
|
||||
"model_evaluation = list(model_evaluations)[0]\n",
|
||||
"print(model_evaluation)"
|
||||
"for model_evaluation in model_evaluations:\n",
|
||||
" print(model_evaluation.to_dict())"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1541,7 +1660,7 @@
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -1685,10 +1804,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" FILE = IMPORT_FILES[0]\n",
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
@@ -1725,10 +1841,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.VideoDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.VideoDataset.create(\n",
|
||||
" display_name=\"human_motion_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.video.classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.video.classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
@@ -1752,9 +1868,7 @@
|
||||
"- `prediction_type`: The type task to train the model for.\n",
|
||||
" - `classification`: A video classification model.\n",
|
||||
" - `object_tracking`: A video object tracking model.\n",
|
||||
" - `action_recognition`: A video action recognition model.\n",
|
||||
"\n",
|
||||
"The instantiated object is the DAG (directed acyclic graph) for the training pipeline."
|
||||
" - `action_recognition`: A video action recognition model."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1765,8 +1879,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLVideoTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
"dag = aiplatform.AutoMLVideoTrainingJob(\n",
|
||||
" display_name=\"human_motion_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
@@ -1781,7 +1895,7 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
@@ -1803,7 +1917,7 @@
|
||||
"source": [
|
||||
"model = dag.run(\n",
|
||||
" dataset=dataset,\n",
|
||||
" model_display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" model_display_name=\"human_motion_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.8,\n",
|
||||
" test_fraction_split=0.2,\n",
|
||||
")"
|
||||
@@ -1816,9 +1930,8 @@
|
||||
},
|
||||
"source": [
|
||||
"## Review model evaluation scores\n",
|
||||
"After your model has finished training, you can review the evaluation scores for it.\n",
|
||||
"\n",
|
||||
"First, you need to get a reference to the new model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project."
|
||||
"After your model training has finished, you can review the evaluation scores for it using the `list_model_evaluations()` method. This method will return an iterator for each evaluation slice."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1829,18 +1942,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"model_evaluations = model.list_model_evaluations()\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
")\n",
|
||||
"model_evaluation = list(model_evaluations)[0]\n",
|
||||
"print(model_evaluation)"
|
||||
"for model_evaluation in model_evaluations:\n",
|
||||
" print(model_evaluation.to_dict())"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1898,16 +2003,7 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1918,66 +2014,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_dataset = True\n",
|
||||
"delete_pipeline = True\n",
|
||||
"delete_model = True\n",
|
||||
"delete_endpoint = True\n",
|
||||
"delete_batchjob = True\n",
|
||||
"delete_customjob = True\n",
|
||||
"delete_hptjob = True\n",
|
||||
"delete_bucket = True\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"# Delete the dataset using the Vertex fully qualified identifier for the dataset\n",
|
||||
"try:\n",
|
||||
" if delete_dataset and \"dataset_id\" in globals():\n",
|
||||
" clients[\"dataset\"].delete_dataset(name=dataset_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n",
|
||||
"try:\n",
|
||||
" if delete_pipeline and \"pipeline_id\" in globals():\n",
|
||||
" clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the model using the Vertex fully qualified identifier for the model\n",
|
||||
"try:\n",
|
||||
" if delete_model and \"model_to_deploy_id\" in globals():\n",
|
||||
" clients[\"model\"].delete_model(name=model_to_deploy_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n",
|
||||
"try:\n",
|
||||
" if delete_endpoint and \"endpoint_id\" in globals():\n",
|
||||
" clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the batch job using the Vertex fully qualified identifier for the batch job\n",
|
||||
"try:\n",
|
||||
" if delete_batchjob and \"batch_job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the custom job using the Vertex fully qualified identifier for the custom job\n",
|
||||
"try:\n",
|
||||
" if delete_customjob and \"job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_custom_job(name=job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n",
|
||||
"try:\n",
|
||||
" if delete_hptjob and \"hpt_job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,14 +33,20 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -67,7 +73,7 @@
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the Penguins dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset predicts the species."
|
||||
"The dataset used for this tutorial is the Penguins dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). This version of the dataset is used to predict the species of penguins from the available features like culmen-length, flipper-depth etc."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -84,16 +90,26 @@
|
||||
"\n",
|
||||
"- `BigQueryML Training`\n",
|
||||
"- `Vertex AI Model resource`\n",
|
||||
"- `Vertex AI Vizier.\n",
|
||||
"- `Vertex AI Vizier`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Create a local BQ table in your project.\n",
|
||||
"- Train a BQML model.\n",
|
||||
"- Evaluate the BQML model.\n",
|
||||
"- Export the BQML model as a cloud model.\n",
|
||||
"- Upload the exported model as a Vertex AI Model resource.\n",
|
||||
"- Hyperparameter tune a BQML model with Vertex AI Vizier."
|
||||
"- Create a local BigQuery table in your project\n",
|
||||
"- Train a BQML model\n",
|
||||
"- Evaluate the BQML model\n",
|
||||
"- Export the BQML model as a cloud model\n",
|
||||
"- Upload the exported model as a `Vertex AI Model` resource\n",
|
||||
"- Hyperparameter tune a BQML model with `Vertex AI Vizier`\n",
|
||||
"- Automatically register a BQML model to `Vertex AI Model Registry`\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"- BigQuery\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -104,7 +120,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -115,19 +131,22 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"# Install the packages\n",
|
||||
"! pip3 install --upgrade pyarrow \\\n",
|
||||
" google-cloud-aiplatform \\\n",
|
||||
" google-cloud-bigquery \\\n",
|
||||
" google-cloud-bigquery-storage $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -165,6 +184,23 @@
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, BigQuery, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,bigquery,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n",
|
||||
"\n",
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
@@ -224,7 +260,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -235,7 +271,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -262,6 +301,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f3bd8c0d0469"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e0953a00668e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -285,7 +385,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -296,8 +397,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -317,7 +419,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -337,7 +439,55 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "set_service_account"
|
||||
},
|
||||
"source": [
|
||||
"#### Service Account\n",
|
||||
"\n",
|
||||
"You use a service account to create Vertex AI Pipeline jobs. If you do not want to use your project's Compute Engine service account, set `SERVICE_ACCOUNT` to another service account ID."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_service_account"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"SERVICE_ACCOUNT = \"[your-service-account]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_service_account"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if (\n",
|
||||
" SERVICE_ACCOUNT == \"\"\n",
|
||||
" or SERVICE_ACCOUNT is None\n",
|
||||
" or SERVICE_ACCOUNT == \"[your-service-account]\"\n",
|
||||
"):\n",
|
||||
" # Get your service account from gcloud\n",
|
||||
" if not IS_COLAB:\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
"\n",
|
||||
" if IS_COLAB:\n",
|
||||
" shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
" project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
" SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -346,9 +496,6 @@
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
@@ -360,28 +507,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"source": [
|
||||
"#### Import BigQuery\n",
|
||||
"\n",
|
||||
"Import the BigQuery package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aiplatform\n",
|
||||
"from google.cloud import bigquery"
|
||||
]
|
||||
},
|
||||
@@ -391,7 +517,7 @@
|
||||
"id": "init_aip:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"### Initialize Vertex AI and BigQuery SDKs for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
@@ -404,7 +530,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -413,8 +539,6 @@
|
||||
"id": "init_bq"
|
||||
},
|
||||
"source": [
|
||||
"### Create BigQuery client\n",
|
||||
"\n",
|
||||
"Create the BigQuery client."
|
||||
]
|
||||
},
|
||||
@@ -426,7 +550,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bqclient = bigquery.Client()"
|
||||
"bqclient = bigquery.Client(project=PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -435,17 +559,17 @@
|
||||
"id": "accelerators:prediction,mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"#### Set hardware accelerators\n",
|
||||
"### Set hardware accelerators\n",
|
||||
"\n",
|
||||
"You can set hardware accelerators for prediction.\n",
|
||||
"\n",
|
||||
"Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n",
|
||||
"\n",
|
||||
" (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
|
||||
" (aiplatform.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
|
||||
"\n",
|
||||
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
|
||||
"\n",
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region"
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -456,13 +580,15 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_DEPLOY_GPU\"):\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_DEPLOY_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (aip.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -471,7 +597,7 @@
|
||||
"id": "container:prediction"
|
||||
},
|
||||
"source": [
|
||||
"#### Set pre-built containers\n",
|
||||
"### Set pre-built containers\n",
|
||||
"\n",
|
||||
"Set the pre-built Docker container image for prediction.\n",
|
||||
"\n",
|
||||
@@ -518,11 +644,11 @@
|
||||
"id": "machine:prediction"
|
||||
},
|
||||
"source": [
|
||||
"#### Set machine type\n",
|
||||
"### Set machine type\n",
|
||||
"\n",
|
||||
"Next, set the machine type to use for prediction.\n",
|
||||
"\n",
|
||||
"- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n",
|
||||
"- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM which is used for prediction.\n",
|
||||
" - `machine type`\n",
|
||||
" - `n1-standard`: 3.75GB of memory per vCPU.\n",
|
||||
" - `n1-highmem`: 6.5GB of memory per vCPU\n",
|
||||
@@ -556,7 +682,7 @@
|
||||
"id": "bqml_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Bigquery ML introduction\n",
|
||||
"## BigQuery ML introduction\n",
|
||||
"\n",
|
||||
"BigQuery ML (BQML) provides the capability to train ML tabular models, such as classification and regression, in BigQuery using SQL syntax.\n",
|
||||
"\n",
|
||||
@@ -581,9 +707,9 @@
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"source": [
|
||||
"### Create BQ dataset/model resource\n",
|
||||
"### Create BQ dataset resource\n",
|
||||
"\n",
|
||||
"First, you create a empty dataset/model resource in your project."
|
||||
"First, you create an empty dataset resource in your project."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -658,7 +784,7 @@
|
||||
"id": "bqml_eval_model"
|
||||
},
|
||||
"source": [
|
||||
"### Evaluate the BQML trained model\n",
|
||||
"### Evaluate the trained BQML model\n",
|
||||
"\n",
|
||||
"Next, retrieve the model evaluation for the trained BQML model.\n",
|
||||
"\n",
|
||||
@@ -693,7 +819,7 @@
|
||||
"source": [
|
||||
"### Export the model from BQML\n",
|
||||
"\n",
|
||||
"The model you trained in BQML is a TensorFlow model. Next, you will export the TensorFlow model artifacts in TF.SavedModel format."
|
||||
"The model you trained in BQML is a TensorFlow model. Next, you export the TensorFlow model artifacts in TF.SavedModel format."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -704,10 +830,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"param = f\"{PROJECT_ID}:{BQ_DATASET_NAME}.{MODEL_NAME} {BUCKET_NAME}/{MODEL_NAME}\"\n",
|
||||
"param = f\"{PROJECT_ID}:{BQ_DATASET_NAME}.{MODEL_NAME} {BUCKET_URI}/{MODEL_NAME}\"\n",
|
||||
"! bq extract -m $param\n",
|
||||
"\n",
|
||||
"MODEL_DIR = f\"{BUCKET_NAME}/{BQ_DATASET_NAME}\"\n",
|
||||
"MODEL_DIR = f\"{BUCKET_URI}/{BQ_DATASET_NAME}\"\n",
|
||||
"! gsutil ls $MODEL_DIR"
|
||||
]
|
||||
},
|
||||
@@ -717,9 +843,25 @@
|
||||
"id": "upload_bqml_model"
|
||||
},
|
||||
"source": [
|
||||
"## Upload the BQML model to a Model resource\n",
|
||||
"## Upload the BigQuery ML model to a Vertex AI Model resource\n",
|
||||
"\n",
|
||||
"Finally, now that you have the BQML model exported as a TF.SavedModel format, you upload the model artifacts to Vertex AI Model resource, in the same way as if you were uploading a custom trained model."
|
||||
"Finally, now that you have the BigQuery ML model exported, you upload the model artifacts to Vertex AI Model resource, in the same way as if you were uploading a custom trained model.\n",
|
||||
"\n",
|
||||
"Below is a partial list of mapping BigQuery ML model types to their corresponding exported model format:\n",
|
||||
"\n",
|
||||
"'LINEAR_REG'<br/>\n",
|
||||
"'LOGISTIC_REG' --> TensorFlow SavedFormat\n",
|
||||
"\n",
|
||||
"'AUTOML_CLASSIFIER'<br/>\n",
|
||||
"'AUTOML_REGRESSOR' --> TensorFlow SavedFormat\n",
|
||||
"\n",
|
||||
"'BOOSTED_TREE_CLASSIFIER'<br/>\n",
|
||||
"'BOOSTED_TREE_REGRESSOR' --> XGBoost format\n",
|
||||
"\n",
|
||||
"'DNN_CLASSIFIER'<br/>\n",
|
||||
"'DNN_REGRESSOR'<br/>\n",
|
||||
"'DNN_LINEAR_COMBINED_CLASSIFIER'<br/>\n",
|
||||
"'DNN_LINEAR_COMBINED_REGRESSOR' --> TensorFlow Estimator"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -730,7 +872,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"model = aip.Model.upload(\n",
|
||||
"model = aiplatform.Model.upload(\n",
|
||||
" display_name=\"penguins_\" + TIMESTAMP,\n",
|
||||
" artifact_uri=MODEL_DIR,\n",
|
||||
" serving_container_image_uri=DEPLOY_IMAGE,\n",
|
||||
@@ -751,7 +893,7 @@
|
||||
"- `deployed_model_display_name`: A human readable name for the deployed model.\n",
|
||||
"- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n",
|
||||
"If only one model, then specify as { \"0\": 100 }, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n",
|
||||
"If there are existing models on the endpoint, for which the traffic will be split, then use model_id to specify as { \"0\": percent, model_id: percent, ... }, where model_id is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n",
|
||||
"If there are existing models on the endpoint, for which the traffic needs to be split, then use model_id to specify as { \"0\": percent, model_id: percent, ... }, where model_id is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n",
|
||||
"- `machine_type`: The type of machine to use for training.\n",
|
||||
"- `accelerator_type`: The hardware accelerator type.\n",
|
||||
"- `accelerator_count`: The number of accelerators to attach to a worker replica.\n",
|
||||
@@ -800,7 +942,7 @@
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -822,9 +964,9 @@
|
||||
"id": "model_delete:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"#### Delete the model\n",
|
||||
"#### Delete the `Vertex AI Model` resource\n",
|
||||
"\n",
|
||||
"The method 'delete()' will delete the model."
|
||||
"The method 'delete()' deletes the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -838,6 +980,32 @@
|
||||
"model.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "7890ae6f6410"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the `BigQuery ML` model\n",
|
||||
"\n",
|
||||
"Next, delete the `BigQuery ML` instance of the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f0b6163e70c0"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_QUERY = f\"\"\"\n",
|
||||
"DROP MODEL `{BQ_DATASET_NAME}.{MODEL_NAME}`\n",
|
||||
"\"\"\"\n",
|
||||
"\n",
|
||||
"job = bqclient.query(MODEL_QUERY)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -846,7 +1014,7 @@
|
||||
"source": [
|
||||
"### Hyperparameter Tune and train a BQML model\n",
|
||||
"\n",
|
||||
"Next, you train a BQML tabular classification model with hyperparameter tuning using the Vertex AI Vizier service. The hyperparameter settings are specified in the `OPTIONS` statement as follows:\n",
|
||||
"Next, you train a BQML tabular classification model with hyperparameter tuning using the `Vertex AI Vizier` service. The hyperparameter settings are specified in the `OPTIONS` statement as follows:\n",
|
||||
"\n",
|
||||
"- `HPARAM_TUNING_ALGORITHM`: The algorithm for selecting the next trial parameters.\n",
|
||||
"- `num_trials`: The number of trials.\n",
|
||||
@@ -901,7 +1069,7 @@
|
||||
"source": [
|
||||
"### Evaluate the BQML trained model\n",
|
||||
"\n",
|
||||
"Next, retrieve the model evaluation for the trained BQML model.\n",
|
||||
"Next, retrieve the model evaluation results for the trained BQML model.\n",
|
||||
"\n",
|
||||
"Learn more about [The ML.EVALUATE function](https://cloud.google.com/bigquery-ml/docs/reference/standard-sql/bigqueryml-syntax-evaluate)."
|
||||
]
|
||||
@@ -926,6 +1094,32 @@
|
||||
"print(results)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3f3cee1236b1"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the `BigQuery ML` model\n",
|
||||
"\n",
|
||||
"Next, delete the `BigQuery ML` instance of the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "957b7d841502"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_QUERY = f\"\"\"\n",
|
||||
"DROP MODEL `{BQ_DATASET_NAME}.{MODEL_NAME}`\n",
|
||||
"\"\"\"\n",
|
||||
"\n",
|
||||
"job = bqclient.query(MODEL_QUERY)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -975,6 +1169,179 @@
|
||||
"print(\"{} created in {}\".format(tblname, job.ended - job.started))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "4def8aaf3398"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the `BigQuery ML` model\n",
|
||||
"\n",
|
||||
"Next, delete the `BigQuery ML` instance of the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ff5b32618018"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_QUERY = f\"\"\"\n",
|
||||
"DROP MODEL `{BQ_DATASET_NAME}.{MODEL_NAME}`\n",
|
||||
"\"\"\"\n",
|
||||
"\n",
|
||||
"job = bqclient.query(MODEL_QUERY)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2b4498ca6fea"
|
||||
},
|
||||
"source": [
|
||||
"## Model Registry\n",
|
||||
"\n",
|
||||
"Alternatively, you can implicitly upload your BigQuery ML model as a `Vertex AI Model` resource with exporting and importing the model artifacts. In this method, you add additional options when training the model that tells BigQuery ML to automatically upload and register the trained model as a `Model` resource.\n",
|
||||
"\n",
|
||||
"### Setting permissions to automatically register the model\n",
|
||||
"\n",
|
||||
"You need to set some additional IAM permissions for BigQuery ML to automatically upload and register the model after training. Depending on your service account, the setting of the permissions below may fail. In this case, we recommend executing the permissions in a Cloud Shell.\n",
|
||||
"\n",
|
||||
"Learn more about [Setting permissions for Model Registry](https://cloud.devsite.corp.google.com/bigquery-ml/docs/managing-models-vertex\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "29229f72d13d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud projects add-iam-policy-binding $PROJECT_ID \\\n",
|
||||
" --member=serviceAccount:$SERVICE_ACCOUNT --role=roles/aiplatform.admin --condition=None"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "6c390ee7c11a"
|
||||
},
|
||||
"source": [
|
||||
"### Training and registering the model\n",
|
||||
"\n",
|
||||
"Next, you train the model and automatically register the model to the `Vertex AI Model Registry`, by adding the following parameters as options:\n",
|
||||
"\n",
|
||||
"- `model_registry`: Set to \"vertex_ai\" to indicate automatic registation to `Vertex AI Model Registry`.\n",
|
||||
"- `vertex_ai_model_id`: The human readable display name for the registered model.\n",
|
||||
"- `vertex_ai_model_version_aliases`: Alternate names for the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "57db464f4c42"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_NAME = \"penguins\"\n",
|
||||
"MODEL_QUERY = f\"\"\"\n",
|
||||
"CREATE OR REPLACE MODEL `{BQ_DATASET_NAME}.{MODEL_NAME}`\n",
|
||||
"OPTIONS(\n",
|
||||
" model_type='DNN_CLASSIFIER',\n",
|
||||
" labels = ['species'],\n",
|
||||
" model_registry=\"vertex_ai\",\n",
|
||||
" vertex_ai_model_id=\"bqml_model_{TIMESTAMP}\", \n",
|
||||
" vertex_ai_model_version_aliases=[\"1\"]\n",
|
||||
" )\n",
|
||||
"AS\n",
|
||||
"SELECT *\n",
|
||||
"FROM `{BQ_TABLE}`\n",
|
||||
"\"\"\"\n",
|
||||
"\n",
|
||||
"job = bqclient.query(MODEL_QUERY)\n",
|
||||
"print(job.errors, job.state)\n",
|
||||
"\n",
|
||||
"while job.running():\n",
|
||||
" from time import sleep\n",
|
||||
"\n",
|
||||
" sleep(30)\n",
|
||||
" print(\"Running ...\")\n",
|
||||
"print(job.errors, job.state)\n",
|
||||
"\n",
|
||||
"tblname = job.ddl_target_table\n",
|
||||
"tblname = \"{}.{}\".format(tblname.dataset_id, tblname.table_id)\n",
|
||||
"print(\"{} created in {}\".format(tblname, job.ended - job.started))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5b4970272040"
|
||||
},
|
||||
"source": [
|
||||
"### Find the model in the `Vertex Model Registry`\n",
|
||||
"\n",
|
||||
"Finally, you can use the `Vertex AI Model` list() method with a filter query to find the automatically registered model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "76c22674ba99"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"models = aiplatform.Model.list(filter=\"display_name=bqml_model_\" + TIMESTAMP)\n",
|
||||
"model = models[0]\n",
|
||||
"\n",
|
||||
"print(model.gca_resource)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "48e6ef5d5ffa"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"models = aiplatform.Model.list()\n",
|
||||
"for model in models:\n",
|
||||
" if model.gca_resource.display_name.startswith(\"bqml\"):\n",
|
||||
" print(model.gca_resource.display_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "ef61354b1a5f"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the `BigQuery ML` model\n",
|
||||
"\n",
|
||||
"Next, delete the `BigQuery ML` instance of the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f6004d1ce59d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_QUERY = f\"\"\"\n",
|
||||
"DROP MODEL `{BQ_DATASET_NAME}.{MODEL_NAME}`\n",
|
||||
"\"\"\"\n",
|
||||
"\n",
|
||||
"job = bqclient.query(MODEL_QUERY)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -986,17 +1353,9 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial.\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"Set `delete_storage` to `True` to delete the Cloud Storage bucket used in this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1007,60 +1366,23 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"# Delete the endpoint using the Vertex endpoint object\n",
|
||||
"endpoint.undeploy_all()\n",
|
||||
"endpoint.delete()\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the model using the Vertex model object\n",
|
||||
"try:\n",
|
||||
" model.delete()\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the created BigQuery dataset\n",
|
||||
"! bq rm -r -f $PROJECT_ID:$BQ_DATASET_NAME\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"delete_storage = False\n",
|
||||
"if delete_storage or os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Delete the created GCS bucket\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -29,18 +29,24 @@
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Logging and Vertex Experiments\n",
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Logging and Vertex AI Experiments\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -56,7 +62,7 @@
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Logging and Vertex Experiments."
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Logging and Vertex AI Experiments."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -93,7 +99,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices for logging data when experimenting or formal training a model.\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are some of the best practices for logging data when experimenting or formally training a model.\n",
|
||||
"\n",
|
||||
"#### Python Logging\n",
|
||||
"\n",
|
||||
@@ -105,7 +111,14 @@
|
||||
"\n",
|
||||
"#### Experiments\n",
|
||||
"\n",
|
||||
"Use Vertex AI Experiments in conjunction with logging when doing experiments to compare results for different experiment configurations."
|
||||
"Use Vertex AI Experiments in conjunction with logging when performing experiments to compare results for different experiment configurations.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -116,7 +129,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -127,19 +140,20 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-logging $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -177,6 +191,24 @@
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, Compute Engine, Cloud Storage and Cloud Logging APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,storage_component,logging).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
@@ -236,7 +268,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -247,7 +279,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -274,6 +309,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f3bd8c0d0469"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e0953a00668e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -283,7 +379,7 @@
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
"### Import libraries"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -294,29 +390,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_logging"
|
||||
},
|
||||
"source": [
|
||||
"#### Import logging\n",
|
||||
"import logging\n",
|
||||
"\n",
|
||||
"Import the logging package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_logging"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import logging"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -338,7 +414,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION)"
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -355,9 +431,9 @@
|
||||
"- Send log output to console.\n",
|
||||
"- Send log output to a file.\n",
|
||||
"\n",
|
||||
"### Logging Levels\n",
|
||||
"### Logging Levels in Python Logging\n",
|
||||
"\n",
|
||||
"The logging levels in order (from least to highest) are, with each level inclusive of the previous level:\n",
|
||||
"The logging levels in order (from least to highest) and each level inclusive of the previous level are :\n",
|
||||
"\n",
|
||||
"1. Informational\n",
|
||||
"2. Warnings\n",
|
||||
@@ -397,7 +473,7 @@
|
||||
"source": [
|
||||
"### Setting logging level\n",
|
||||
"\n",
|
||||
"To set the logging level, you get the logging handler using `getLogger()`. You can have multiple logging handles. When `getLogger()` is called w/o arguments it gets the default handler, named ROOT. With the handler, you set the logging level with the method 'setLevel()`."
|
||||
"To set the logging level, you get the logging handler using `getLogger()`. You can have multiple logging handles. When `getLogger()` is called without any arguments, it gets the default handler named ROOT. With the handler, you set the logging level with the method `setLevel()`."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -444,7 +520,7 @@
|
||||
"source": [
|
||||
"### Output to a local file\n",
|
||||
"\n",
|
||||
"You can preserve your logging output to a file that is local to where the Python script is running with the method `BasicConfig()`, with the following paraneters:\n",
|
||||
"You can preserve your logging output to a file that is local to where the Python script is running with the method `BasicConfig()`, that takes the following parameters:\n",
|
||||
"\n",
|
||||
"- `filename`: The file path to the local file to write the log output to.\n",
|
||||
"- `level`: Sets the level of logging that is written to the logging file.\n",
|
||||
@@ -481,7 +557,7 @@
|
||||
"- Send log output to storage.\n",
|
||||
"- Retrieve log output from storage.\n",
|
||||
"\n",
|
||||
"### Logging Levels\n",
|
||||
"### Logging Levels in Cloud Logging\n",
|
||||
"\n",
|
||||
"The logging levels in order (from least to highest) are, with each level inclusive of the previous level:\n",
|
||||
"\n",
|
||||
@@ -516,7 +592,7 @@
|
||||
"from google.cloud.logging.handlers import CloudLoggingHandler\n",
|
||||
"\n",
|
||||
"# Connect to the Cloud Logging service\n",
|
||||
"cl_client = google.cloud.logging.Client()\n",
|
||||
"cl_client = google.cloud.logging.Client(project=PROJECT_ID)\n",
|
||||
"handler = CloudLoggingHandler(cl_client, name=\"mylog\")\n",
|
||||
"\n",
|
||||
"# Create a logger instance and logging level\n",
|
||||
@@ -538,7 +614,7 @@
|
||||
"source": [
|
||||
"### Logging output\n",
|
||||
"\n",
|
||||
"To log output at specific levels is identical in method, and method names, as in Python logging, except that you use your instance of the cloud logger in place of logging."
|
||||
"Logging output at specific levels is identical to Python logging with respect to method and method names. The only difference is that you use your instance of the cloud logger in place of logging."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -566,7 +642,7 @@
|
||||
"To get the logged output, you:\n",
|
||||
"\n",
|
||||
"1. Retrieve the log handle to the service.\n",
|
||||
"2. Using the handle call the method `list_entries()`\n",
|
||||
"2. Using the handle, call the method `list_entries()`.\n",
|
||||
"3. Iterate through the entries."
|
||||
]
|
||||
},
|
||||
@@ -593,10 +669,10 @@
|
||||
"source": [
|
||||
"## Logging with Vertex AI Experiments and Vertex AI ML Metadata\n",
|
||||
"\n",
|
||||
"You can log results related to training experiments with `Vertex AI Experiments` and `ML Metadata`:\n",
|
||||
"You can log results related to training experiments with `Vertex AI Experiments` and `ML Metadata` including:\n",
|
||||
"\n",
|
||||
"- Preserve results of an experiment.\n",
|
||||
"- Track multiple runs -- i.e., training runs -- within an experiment.\n",
|
||||
"- Track multiple runs i.e., training runs within an experiment.\n",
|
||||
"- Track parameters (configuration) and metrics (results).\n",
|
||||
"- Retrieve and display the logged output.\n",
|
||||
"\n",
|
||||
@@ -611,14 +687,29 @@
|
||||
"source": [
|
||||
"### Create experiment for tracking training related metadata\n",
|
||||
"\n",
|
||||
"Setup tracking the parameters (configuration) and metrics (results) for each experiment:\n",
|
||||
"Setup tracking for parameters (configuration) and metrics (results) in each experiment:\n",
|
||||
"\n",
|
||||
"- `aip.init()` - Create an experiment instance\n",
|
||||
"- `aip.start_run()` - Track a specific run within the experiment.\n",
|
||||
"- `aiplatform.init()` - Create an experiment instance\n",
|
||||
"- `aiplatform.start_run()` - Track a specific run within the experiment.\n",
|
||||
"\n",
|
||||
"Learn more about [Introduction to Vertex AI ML Metadata](https://cloud.google.com/vertex-ai/docs/ml-metadata/introduction)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "1ed46e349cf2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Specify a name for the experiment\n",
|
||||
"EXPERIMENT_NAME = \"[your-experiment-name]\"\n",
|
||||
"\n",
|
||||
"if EXPERIMENT_NAME == \"[your-experiment-name]\":\n",
|
||||
" EXPERIMENT_NAME = \"example-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -627,9 +718,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPERIMENT_NAME = \"example-\" + TIMESTAMP\n",
|
||||
"aip.init(experiment=EXPERIMENT_NAME)\n",
|
||||
"aip.start_run(\"run-1\")"
|
||||
"# Create experiment\n",
|
||||
"aiplatform.init(experiment=EXPERIMENT_NAME)\n",
|
||||
"aiplatform.start_run(\"run-1\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -640,14 +731,14 @@
|
||||
"source": [
|
||||
"### Log parameters for the experiment\n",
|
||||
"\n",
|
||||
"Typically, an experiment is associated with a specific dataset and model architecture. Within an experiment, you may have multiple training runs, where each run tries a different configuration. As examples:\n",
|
||||
"Typically, an experiment is associated with a specific dataset and a model architecture. Within an experiment, you may have multiple training runs, where each run tries a different configuration. For example:\n",
|
||||
"\n",
|
||||
"- Dataset split\n",
|
||||
"- Dataset sampling and boosting\n",
|
||||
"- Depth and width of layers\n",
|
||||
"- Hyperparameters\n",
|
||||
"\n",
|
||||
"These configuration settings are referred to as parameters, which you store their key/value pair using the method `log_params()`"
|
||||
"These configuration settings are referred to as parameters, which you store as key-value pairs using the method `log_params()`"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -662,7 +753,7 @@
|
||||
"hyperparams[\"epochs\"] = 100\n",
|
||||
"hyperparams[\"batch_size\"] = 32\n",
|
||||
"hyperparams[\"learning_rate\"] = 0.01\n",
|
||||
"aip.log_params(hyperparams)"
|
||||
"aiplatform.log_params(hyperparams)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -673,14 +764,14 @@
|
||||
"source": [
|
||||
"### Log metrics for the experiment\n",
|
||||
"\n",
|
||||
"At the completion, or termination, of a run within an experiment, you can log results that you use to compare runs. As examples:\n",
|
||||
"At the completion or termination of a run within an experiment, you can log results that you use to compare runs. For example:\n",
|
||||
"\n",
|
||||
"- Evaluation metrics\n",
|
||||
"- Hyperparameter search selection\n",
|
||||
"- Time to train the model\n",
|
||||
"- Early stop trigger\n",
|
||||
"\n",
|
||||
"These results settings are referred to as metrics, which you store their key/value pair using the method `log_metrics()`"
|
||||
"These results are referred to as metrics, which you store as key-value pairs using the method `log_metrics()`"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -694,7 +785,7 @@
|
||||
"metrics = {}\n",
|
||||
"metrics[\"test_acc\"] = 98.7\n",
|
||||
"metrics[\"train_acc\"] = 99.3\n",
|
||||
"aip.log_metrics(metrics)"
|
||||
"aiplatform.log_metrics(metrics)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -716,8 +807,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"experiment_df = aip.get_experiment_df()\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == \"example\"]\n",
|
||||
"experiment_df = aiplatform.get_experiment_df()\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == EXPERIMENT_NAME]\n",
|
||||
"experiment_df.T"
|
||||
]
|
||||
},
|
||||
@@ -734,15 +825,9 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"### Delete the experiment\n",
|
||||
"\n",
|
||||
"Next, delete the experiment. You will need to get the context via the metadata to delete it."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -753,60 +838,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"c = aiplatform.metadata._Context(EXPERIMENT_NAME)\n",
|
||||
"c.delete()"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -29,18 +29,24 @@
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex Tensorboard\n",
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex AI Tensorboard\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/notebook_template.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -56,7 +62,7 @@
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Vertex Tensorboard."
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Vertex AI Tensorboard."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -81,6 +87,68 @@
|
||||
"- Using Vertex AI TensorBoard with Vertex AI Training."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "b132d4ef86d6"
|
||||
},
|
||||
"source": [
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "56cb7f08a9e8"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your local development environment\n",
|
||||
"\n",
|
||||
"**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n",
|
||||
"all the requirements to run this notebook. You can skip this step.\n",
|
||||
"\n",
|
||||
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
|
||||
"You need the following:\n",
|
||||
"\n",
|
||||
"* The Google Cloud SDK\n",
|
||||
"* Git\n",
|
||||
"* Python 3\n",
|
||||
"* virtualenv\n",
|
||||
"* Jupyter notebook running in a virtual environment with Python 3\n",
|
||||
"\n",
|
||||
"The Google Cloud guide to [Setting up a Python development\n",
|
||||
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
|
||||
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
|
||||
"for meeting these requirements. The following steps provide a condensed set of\n",
|
||||
"instructions:\n",
|
||||
"\n",
|
||||
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
|
||||
"\n",
|
||||
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
|
||||
"\n",
|
||||
"1. [Install\n",
|
||||
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
|
||||
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
|
||||
"\n",
|
||||
"1. To install Jupyter, run `pip3 install jupyter` on the\n",
|
||||
"command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"1. Open this notebook in the Jupyter Notebook Dashboard.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -89,7 +157,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices for visualizing your training with TensorBoard.\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are the best practices for visualizing your training with TensorBoard.\n",
|
||||
"\n",
|
||||
"#### Local TensorBoard\n",
|
||||
"\n",
|
||||
@@ -112,30 +180,32 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
"id": "020040f91150"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install -U tensorflow==2.8 $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -167,6 +237,32 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2721ef0202d9"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com). \n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -232,7 +328,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -243,7 +339,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -270,6 +369,82 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2700e693f1b3"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
|
||||
"authenticated. Skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "885395904904"
|
||||
},
|
||||
"source": [
|
||||
"**If you are using Colab**, run the cell below and follow the instructions\n",
|
||||
"when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"1. In the Cloud Console, go to the [**Create service account key**\n",
|
||||
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
|
||||
"\n",
|
||||
"2. Click **Create service account**.\n",
|
||||
"\n",
|
||||
"3. In the **Service account name** field, enter a name, and\n",
|
||||
" click **Create**.\n",
|
||||
"\n",
|
||||
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
|
||||
"into the filter box, and select\n",
|
||||
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
|
||||
"local environment.\n",
|
||||
"\n",
|
||||
"6. Enter the path to your service account key as the\n",
|
||||
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "eff327d0552b"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -293,7 +468,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -304,8 +479,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -325,7 +500,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -345,7 +520,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -383,9 +558,16 @@
|
||||
" or SERVICE_ACCOUNT is None\n",
|
||||
" or SERVICE_ACCOUNT == \"[your-service-account]\"\n",
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" # Get your service account from gcloud\n",
|
||||
" if not IS_COLAB:\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
"\n",
|
||||
" if IS_COLAB:\n",
|
||||
" shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
" project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
" SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -409,7 +591,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -453,7 +635,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -483,13 +665,15 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (aip.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
" TRAIN_GPU, TRAIN_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -662,9 +846,9 @@
|
||||
"\n",
|
||||
"You can upload your TensorBoard logs and share with others using `tensorboard dev` command. Once uploaded, a URL is returned to open up the TensorBoard instance in a brower for visualizing.\n",
|
||||
"\n",
|
||||
"*Note:* Your TensorBoard instance is publicly visable.\n",
|
||||
"*Note:* Your TensorBoard instance is publicly visible.\n",
|
||||
"\n",
|
||||
"*Note:* In this example, while running within a notebook, the command will freeze since it is waiting for an interactive yes/no input. You can kill the command with a Ctrl C or kernel interupt.\n",
|
||||
"*Note:* This cell is for demonstration purposes and must be ran in a terminal shell. In this example, while running within a notebook, the command will freeze since it is waiting for an interactive yes/no input. You can kill the command with a Ctrl C or kernel interupt.\n",
|
||||
"\n",
|
||||
"Learn more about [What is TensorBoard.dev](https://tensorboard.dev/)."
|
||||
]
|
||||
@@ -677,7 +861,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! tensorboard dev upload --logdir {LOG_DIR} \\\n",
|
||||
"! tensorboard dev upload --logdir logs \\\n",
|
||||
" --name \"Simple experiment with MNIST\" \\\n",
|
||||
" --description \"Training results\" \\\n",
|
||||
" --one_shot"
|
||||
@@ -705,7 +889,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"TENSORBOARD_DISPLAY_NAME = \"example\"\n",
|
||||
"tensorboard = aip.Tensorboard.create(display_name=TENSORBOARD_DISPLAY_NAME)\n",
|
||||
"tensorboard = aiplatform.Tensorboard.create(display_name=TENSORBOARD_DISPLAY_NAME)\n",
|
||||
"tensorboard_resource_name = tensorboard.gca_resource.name\n",
|
||||
"print(\"TensorBoard resource name:\", tensorboard_resource_name)"
|
||||
]
|
||||
@@ -745,9 +929,9 @@
|
||||
"\n",
|
||||
"url = output[1].split(' ')[-1]\n",
|
||||
"\n",
|
||||
"print(url)\n",
|
||||
"#print(url)\n",
|
||||
"\n",
|
||||
"from IPython.core.display import display, HTML\n",
|
||||
"from IPython.display import display, HTML\n",
|
||||
"display(HTML(\"<a href='\" + url + \"'>click here for TensorBoard instance</a>\"))"
|
||||
]
|
||||
},
|
||||
@@ -952,7 +1136,7 @@
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_example.tar.gz"
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_example.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -984,7 +1168,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job = aip.CustomTrainingJob(\n",
|
||||
"job = aiplatform.CustomTrainingJob(\n",
|
||||
" display_name=\"example_\" + TIMESTAMP,\n",
|
||||
" script_path=\"custom/trainer/task.py\",\n",
|
||||
" container_uri=TRAIN_IMAGE,\n",
|
||||
@@ -1020,7 +1204,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, TIMESTAMP)\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, TIMESTAMP)\n",
|
||||
"\n",
|
||||
"EPOCHS = 20\n",
|
||||
"STEPS = 100\n",
|
||||
@@ -1107,6 +1291,28 @@
|
||||
"Alternatively, you can navigate to the Experiments tab and view the list of all experiments. Your experiment will have the same name as the training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_tensorboard"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the TensorBoard instance\n",
|
||||
"\n",
|
||||
"Next, delete the TensorBoard instance."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_tensorboard"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"tensorboard.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -1120,14 +1326,8 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -1139,60 +1339,14 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"# Delete the custom training job\n",
|
||||
"job.delete()\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -29,18 +29,24 @@
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex Training\n",
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex AI Training\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -127,7 +133,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the packages required for executing this notebook"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -138,19 +144,20 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Vertex AI Workbench Notebook product has specific requirements\n",
|
||||
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
|
||||
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
|
||||
" \"/opt/deeplearning/metadata/env_version\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_WORKBENCH_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -172,8 +179,6 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Automatically restart kernel after installs\n",
|
||||
" import IPython\n",
|
||||
@@ -182,6 +187,36 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "before_you_begin"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### GPU runtime\n",
|
||||
"\n",
|
||||
"*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n",
|
||||
"\n",
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
|
||||
"\n",
|
||||
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,storage-component.googleapis.com)\n",
|
||||
"\n",
|
||||
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
|
||||
"\n",
|
||||
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -247,7 +282,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -258,7 +293,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -285,6 +323,67 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "gcp_authenticate"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"**Click Create service account**.\n",
|
||||
"\n",
|
||||
"In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "gcp_authenticate"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Vertex AI Workbench, then don't execute this code\n",
|
||||
"IS_COLAB = False\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
|
||||
" \"DL_ANACONDA_HOME\"\n",
|
||||
"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" IS_COLAB = True\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -308,7 +407,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -319,8 +419,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -340,7 +441,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -360,7 +461,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -383,7 +484,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -405,7 +506,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -425,9 +526,9 @@
|
||||
"\n",
|
||||
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
|
||||
"\n",
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region\n",
|
||||
"Learn more about [hardware accelerator support for your region](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators).\n",
|
||||
"\n",
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3. This is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -440,7 +541,7 @@
|
||||
"source": [
|
||||
"if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
@@ -448,7 +549,7 @@
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_DEPLOY_GPU\"):\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_DEPLOY_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
@@ -483,7 +584,7 @@
|
||||
"if os.getenv(\"IS_TESTING_TF\"):\n",
|
||||
" TF = os.getenv(\"IS_TESTING_TF\")\n",
|
||||
"else:\n",
|
||||
" TF = \"2.1\".replace(\".\", \"-\")\n",
|
||||
" TF = \"2.5\".replace(\".\", \"-\")\n",
|
||||
"\n",
|
||||
"if TF[0] == \"2\":\n",
|
||||
" if TRAIN_GPU:\n",
|
||||
@@ -601,7 +702,7 @@
|
||||
"DISPLAY_NAME = \"boston_\" + TIMESTAMP\n",
|
||||
"REQUIREMENTS = [\"tensorflow==2.3\"]\n",
|
||||
"\n",
|
||||
"job = aip.CustomTrainingJob(\n",
|
||||
"job = aiplatform.CustomTrainingJob(\n",
|
||||
" display_name=DISPLAY_NAME,\n",
|
||||
" script_path=\"task.py\",\n",
|
||||
" requirements=REQUIREMENTS,\n",
|
||||
@@ -622,7 +723,7 @@
|
||||
"In summary:\n",
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Write \"hello world\" to the file."
|
||||
]
|
||||
},
|
||||
@@ -692,12 +793,12 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"CMDARGS = [\n",
|
||||
" \"--model-dir=\" + BUCKET_NAME,\n",
|
||||
" \"--model-dir=\" + BUCKET_URI,\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"job.run(args=CMDARGS, replica_count=1, machine_type=TRAIN_COMPUTE, sync=True)\n",
|
||||
"\n",
|
||||
"! gsutil cat {BUCKET_NAME}/test.txt"
|
||||
"! gsutil cat {BUCKET_URI}/test.txt"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -767,9 +868,9 @@
|
||||
"source": [
|
||||
"DISPLAY_NAME = \"boston_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"job = aip.CustomPythonPackageTrainingJob(\n",
|
||||
"job = aiplatform.CustomPythonPackageTrainingJob(\n",
|
||||
" display_name=DISPLAY_NAME,\n",
|
||||
" python_package_gcs_uri=f\"{BUCKET_NAME}/trainer_boston.tar.gz\",\n",
|
||||
" python_package_gcs_uri=f\"{BUCKET_URI}/trainer_boston.tar.gz\",\n",
|
||||
" python_module_name=\"trainer.task\",\n",
|
||||
" container_uri=TRAIN_IMAGE,\n",
|
||||
")"
|
||||
@@ -847,7 +948,7 @@
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Get the number of epochs to run from the command line (`--model_dir`).\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Repeat appending \"hello world\" to the file, one per epoch."
|
||||
]
|
||||
},
|
||||
@@ -899,7 +1000,7 @@
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz"
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_boston.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -921,11 +1022,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"CMDARGS = [\"--model-dir=\" + BUCKET_NAME, \"--epochs=5\"]\n",
|
||||
"CMDARGS = [\"--model-dir=\" + BUCKET_URI, \"--epochs=5\"]\n",
|
||||
"\n",
|
||||
"job.run(args=CMDARGS, replica_count=1, machine_type=TRAIN_COMPUTE, sync=True)\n",
|
||||
"\n",
|
||||
"! gsutil cat {BUCKET_NAME}/test.txt"
|
||||
"! gsutil cat {BUCKET_URI}/test.txt"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1044,7 +1145,7 @@
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Get the number of epochs to run from the command line (`--model_dir`).\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Repeat appending \"hello world\" to the file, one per epoch."
|
||||
]
|
||||
},
|
||||
@@ -1150,7 +1251,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! docker build custom -t $TRAIN_IMAGE"
|
||||
"if not IS_COLAB:\n",
|
||||
" ! docker build custom -t $TRAIN_IMAGE\n",
|
||||
"else:\n",
|
||||
" # install docker daemon\n",
|
||||
" ! apt-get -qq install docker.io"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1172,7 +1277,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! docker run $TRAIN_IMAGE --epochs=5 --model-dir=./"
|
||||
"if not IS_COLAB:\n",
|
||||
" ! docker run $TRAIN_IMAGE --epochs=5 --model-dir=./"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1194,7 +1300,38 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! docker push $TRAIN_IMAGE"
|
||||
"if not IS_COLAB:\n",
|
||||
" ! docker push $TRAIN_IMAGE"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f50e9c553fb7"
|
||||
},
|
||||
"source": [
|
||||
"*Executes in Colab*"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a7e8c98f1e56"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%bash -s $IS_COLAB $TRAIN_IMAGE\n",
|
||||
"if [ $1 == \"False\" ]; then\n",
|
||||
" exit 0\n",
|
||||
"fi\n",
|
||||
"set -x\n",
|
||||
"dockerd -b none --iptables=0 -l warn &\n",
|
||||
"for i in $(seq 5); do [ ! -S \"/var/run/docker.sock\" ] && sleep 2 || break; done\n",
|
||||
"docker build custom -t $2\n",
|
||||
"docker run $2 --epochs=5 --model-dir=./\n",
|
||||
"docker push $2\n",
|
||||
"kill $(jobs -p)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1228,7 +1365,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job = aip.CustomContainerTrainingJob(\n",
|
||||
"job = aiplatform.CustomContainerTrainingJob(\n",
|
||||
" display_name=\"boston_\" + TIMESTAMP,\n",
|
||||
" container_uri=TRAIN_IMAGE,\n",
|
||||
" command=[\"python3\", \"trainer/task.py\"],\n",
|
||||
@@ -1256,11 +1393,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"CMDARGS = [\"--model-dir=\" + BUCKET_NAME, \"--epochs=5\"]\n",
|
||||
"CMDARGS = [\"--model-dir=\" + BUCKET_URI, \"--epochs=5\"]\n",
|
||||
"\n",
|
||||
"job.run(args=CMDARGS, replica_count=1, machine_type=TRAIN_COMPUTE, sync=True)\n",
|
||||
"\n",
|
||||
"! gsutil cat {BUCKET_NAME}/test.txt"
|
||||
"! gsutil cat {BUCKET_URI}/test.txt"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1465,7 +1602,7 @@
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz"
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_boston.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1495,7 +1632,7 @@
|
||||
"if os.getenv(\"IS_TESTING_TF\"):\n",
|
||||
" TF = os.getenv(\"IS_TESTING_TF\")\n",
|
||||
"else:\n",
|
||||
" TF = \"2.1\".replace(\".\", \"-\")\n",
|
||||
" TF = \"2.5\".replace(\".\", \"-\")\n",
|
||||
"\n",
|
||||
"if TF[0] == \"2\":\n",
|
||||
" if TRAIN_GPU:\n",
|
||||
@@ -1550,9 +1687,9 @@
|
||||
"source": [
|
||||
"DISPLAY_NAME = \"boston_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"job = aip.CustomPythonPackageTrainingJob(\n",
|
||||
"job = aiplatform.CustomPythonPackageTrainingJob(\n",
|
||||
" display_name=DISPLAY_NAME,\n",
|
||||
" python_package_gcs_uri=f\"{BUCKET_NAME}/trainer_boston.tar.gz\",\n",
|
||||
" python_package_gcs_uri=f\"{BUCKET_URI}/trainer_boston.tar.gz\",\n",
|
||||
" python_module_name=\"trainer.task\",\n",
|
||||
" container_uri=TRAIN_IMAGE,\n",
|
||||
" model_serving_container_image_uri=DEPLOY_IMAGE,\n",
|
||||
@@ -1586,12 +1723,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, TIMESTAMP)\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, TIMESTAMP)\n",
|
||||
"\n",
|
||||
"EPOCHS = 20\n",
|
||||
"STEPS = 100\n",
|
||||
"\n",
|
||||
"DIRECT = True\n",
|
||||
"DIRECT = False\n",
|
||||
"if DIRECT:\n",
|
||||
" CMDARGS = [\n",
|
||||
" \"--model-dir=\" + MODEL_DIR,\n",
|
||||
@@ -1757,17 +1894,7 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1778,60 +1905,24 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"delete_bucket = False\n",
|
||||
"delete_model = True\n",
|
||||
"delete_job = True\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
"if delete_model:\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
"if delete_job:\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -rf {BUCKET_URI}"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||