mirror of
https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
synced 2026-09-26 14:42:04 +00:00
Compare commits
472
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6be5abbc25 | ||
|
|
912a52de70 | ||
|
|
5e29090e86 | ||
|
|
cd8fcd1839 | ||
|
|
e51075ec4b | ||
|
|
9709c0dddb | ||
|
|
a21ae41762 | ||
|
|
ff5ff7b609 | ||
|
|
23748f443e | ||
|
|
fc6b2167de | ||
|
|
e79a45358c | ||
|
|
20d19fb11c | ||
|
|
9c9f7a6e2a | ||
|
|
ede41c2115 | ||
|
|
ca53786c04 | ||
|
|
419f8310c9 | ||
|
|
996b690e03 | ||
|
|
5b9d04d63a | ||
|
|
9ed3c2f83d | ||
|
|
5fc93c8bdb | ||
|
|
3cf46226e9 | ||
|
|
439f6a0cae | ||
|
|
993898bb71 | ||
|
|
9e9e639375 | ||
|
|
633cf6a799 | ||
|
|
8b618bc455 | ||
|
|
6bbe3bcfe0 | ||
|
|
814827ac19 | ||
|
|
7a613785b9 | ||
|
|
100243e90a | ||
|
|
6424515b03 | ||
|
|
8d22b221b4 | ||
|
|
2a8ad7cdbb | ||
|
|
3239b301f2 | ||
|
|
acb10d14b8 | ||
|
|
778d145970 | ||
|
|
4d00356f4b | ||
|
|
0654305994 | ||
|
|
c53f392c5a | ||
|
|
f4b56e92ae | ||
|
|
45ec1cf18a | ||
|
|
f6c8bcf937 | ||
|
|
9e590d5a9f | ||
|
|
46e0ea4f1c | ||
|
|
300fce6b9f | ||
|
|
52e3066c38 | ||
|
|
27ebf52198 | ||
|
|
ca7d4e153e | ||
|
|
5b6c766629 | ||
|
|
5efa51206f | ||
|
|
e604a4d43e | ||
|
|
23af5373ec | ||
|
|
7577c0b1fc | ||
|
|
b648f9e73b | ||
|
|
966bbc49a7 | ||
|
|
f5d341ae45 | ||
|
|
70770a50c7 | ||
|
|
f181c39cbf | ||
|
|
a5637f87f2 | ||
|
|
0103299084 | ||
|
|
1867536d76 | ||
|
|
0edae683e7 | ||
|
|
8471b5cb6f | ||
|
|
87f540ac53 | ||
|
|
babeba9f02 | ||
|
|
85c649dd26 | ||
|
|
b7135ae1f0 | ||
|
|
2f5119a266 | ||
|
|
23e64ca76f | ||
|
|
0ba5a62cc9 | ||
|
|
0be2c6fd0c | ||
|
|
cef4928c49 | ||
|
|
9bb8107110 | ||
|
|
b075990d88 | ||
|
|
6a83c4c695 | ||
|
|
820c0f8db4 | ||
|
|
4ab197a4ba | ||
|
|
2a5877fbd1 | ||
|
|
090e1d9fee | ||
|
|
2e049d4830 | ||
|
|
e3320d2126 | ||
|
|
5fc0e03ca3 | ||
|
|
ff2a16237d | ||
|
|
d26f081642 | ||
|
|
8a0a39176c | ||
|
|
646532ea69 | ||
|
|
9e96a3da67 | ||
|
|
db34e1fbd5 | ||
|
|
82308acbac | ||
|
|
ee0ba75d1e | ||
|
|
d61aedc721 | ||
|
|
19f7f94af5 | ||
|
|
4e5ce9b226 | ||
|
|
99938244f4 | ||
|
|
0cc7be4a6a | ||
|
|
b6bde41850 | ||
|
|
52444a0933 | ||
|
|
bab9c398fd | ||
|
|
447affcc93 | ||
|
|
6132c37be0 | ||
|
|
8d7f59aeec | ||
|
|
bd327ad424 | ||
|
|
3b1fbdb382 | ||
|
|
954043a729 | ||
|
|
16ef9ee80e | ||
|
|
79301b4a4d | ||
|
|
0b38d02e6f | ||
|
|
c52ff25ba4 | ||
|
|
424400bace | ||
|
|
b1dfac2043 | ||
|
|
b81ffcddab | ||
|
|
21d8f144aa | ||
|
|
065a674305 | ||
|
|
571d498d08 | ||
|
|
e936882123 | ||
|
|
f754f99052 | ||
|
|
aa5523a5e9 | ||
|
|
81393ede1a | ||
|
|
5a1c0222da | ||
|
|
1f9326bd56 | ||
|
|
a94cae2e79 | ||
|
|
cb861713c8 | ||
|
|
3a55087789 | ||
|
|
d359b21f3e | ||
|
|
c7d4123b25 | ||
|
|
07a8bb2d0c | ||
|
|
f6b6f365b6 | ||
|
|
fec825f9e5 | ||
|
|
1e9bf72097 | ||
|
|
3fa1cf99ec | ||
|
|
2bcaf8abde | ||
|
|
759495a1f8 | ||
|
|
9925e62c4b | ||
|
|
40ade71b35 | ||
|
|
65173071a5 | ||
|
|
a514bb51c2 | ||
|
|
a6dc1b0f6d | ||
|
|
75236998c7 | ||
|
|
9c8a7808bf | ||
|
|
86ce1576d2 | ||
|
|
c2b743bfeb | ||
|
|
b04575d746 | ||
|
|
b90d16885e | ||
|
|
828f1a26fa | ||
|
|
915e5edf0a | ||
|
|
5d4c1be285 | ||
|
|
275bfb8f69 | ||
|
|
93cb0ca3d9 | ||
|
|
41d60d052d | ||
|
|
d872cdcf7e | ||
|
|
ada5e4a854 | ||
|
|
ecb32b099d | ||
|
|
f8d09e8e9b | ||
|
|
007df88fba | ||
|
|
c70f3ef9a3 | ||
|
|
505e101452 | ||
|
|
b77d51b58b | ||
|
|
deaa1ccf2b | ||
|
|
f6e38860aa | ||
|
|
4937e382b1 | ||
|
|
0fe2770947 | ||
|
|
80d7ee67d6 | ||
|
|
f17e2d6c8d | ||
|
|
ae1cd0ed08 | ||
|
|
a68b491edc | ||
|
|
d39bed012a | ||
|
|
f8c93de1c4 | ||
|
|
0b5d97aa3a | ||
|
|
87f0785ed1 | ||
|
|
09c8ea616b | ||
|
|
6c11241cf7 | ||
|
|
9b7dc2e4fd | ||
|
|
9b3d43b1a6 | ||
|
|
57ee4e2eab | ||
|
|
4d736d7992 | ||
|
|
69e92a650d | ||
|
|
4ed979eec4 | ||
|
|
0344de8090 | ||
|
|
83eca0f0cc | ||
|
|
4619b272fa | ||
|
|
091fa33c01 | ||
|
|
42a05c2b35 | ||
|
|
8ac32fa42f | ||
|
|
0303057f11 | ||
|
|
7ae13b346a | ||
|
|
44390cbd99 | ||
|
|
58dab0b1bb | ||
|
|
8d36834fcd | ||
|
|
31ca3e36f4 | ||
|
|
a72d7dc49f | ||
|
|
9efbd48233 | ||
|
|
166f0f8ce7 | ||
|
|
bd9f9675cf | ||
|
|
42ec0e6ad9 | ||
|
|
fde3d98a2a | ||
|
|
b83f869a44 | ||
|
|
b918768776 | ||
|
|
a22aceb10c | ||
|
|
c8b53a614e | ||
|
|
7be7d5be44 | ||
|
|
02a030d6f1 | ||
|
|
fc8e9e4483 | ||
|
|
fa019e051a | ||
|
|
01d513eea3 | ||
|
|
928822cc0c | ||
|
|
1dfb4091ea | ||
|
|
ffb5ea7e88 | ||
|
|
dd2028d76c | ||
|
|
70daf2e605 | ||
|
|
42db3643d0 | ||
|
|
66ce2fa5b7 | ||
|
|
44a63c8186 | ||
|
|
a29f7a376b | ||
|
|
66667ea1db | ||
|
|
0588a7b62d | ||
|
|
ba74664aa4 | ||
|
|
26011283d2 | ||
|
|
737c635a59 | ||
|
|
e998333f34 | ||
|
|
6ea2b5dd24 | ||
|
|
48f9d563a4 | ||
|
|
9167c42cbc | ||
|
|
9bd261b2c0 | ||
|
|
12657051bd | ||
|
|
b9471efe45 | ||
|
|
01d8d165da | ||
|
|
de28ab9241 | ||
|
|
52b91234e9 | ||
|
|
eaa827959f | ||
|
|
489bdfc092 | ||
|
|
2631ce8f3b | ||
|
|
cb3fcaa27e | ||
|
|
6a971663b3 | ||
|
|
e29e127d7b | ||
|
|
0ff91f926d | ||
|
|
ca19b8f8e7 | ||
|
|
ad06948c12 | ||
|
|
7ed8313d37 | ||
|
|
c559851f78 | ||
|
|
7374698440 | ||
|
|
b77fb73ad0 | ||
|
|
f5e9d7a9ed | ||
|
|
e9c0a56b72 | ||
|
|
d4545dbc61 | ||
|
|
d22db6d795 | ||
|
|
45af74953a | ||
|
|
fd5574fa12 | ||
|
|
85fa955e6d | ||
|
|
bbed90a494 | ||
|
|
a037d2bd78 | ||
|
|
5de7f31c07 | ||
|
|
c8b50b4195 | ||
|
|
4589293efc | ||
|
|
bdf5745870 | ||
|
|
a4b5c22aa2 | ||
|
|
ecfddc0edc | ||
|
|
385a8ca2ea | ||
|
|
0e91687156 | ||
|
|
d04c79b378 | ||
|
|
ff8a9b9ac5 | ||
|
|
5676d07dbd | ||
|
|
be1f7aa631 | ||
|
|
630de0bea8 | ||
|
|
e7da210369 | ||
|
|
c57dd78a86 | ||
|
|
4b5fe2c3cc | ||
|
|
92e0c1b1c7 | ||
|
|
e563a66114 | ||
|
|
de13d8c65c | ||
|
|
a72b7bc9b0 | ||
|
|
e1cd3ce080 | ||
|
|
c18f3754e1 | ||
|
|
844bfbd2d6 | ||
|
|
a9a512cdc4 | ||
|
|
51b2be8ab4 | ||
|
|
ca52e70dcf | ||
|
|
2fe032efcb | ||
|
|
1f3a7418f4 | ||
|
|
afacbe8ba6 | ||
|
|
74376f2787 | ||
|
|
abc0e9dec4 | ||
|
|
d0fb60fb7d | ||
|
|
7d4fb0ff8d | ||
|
|
a6e69a4561 | ||
|
|
fdaa5a6b90 | ||
|
|
5ba56dfc71 | ||
|
|
4c04abe724 | ||
|
|
67a2f84f6e | ||
|
|
53d58f16a4 | ||
|
|
f27aec1295 | ||
|
|
559e476170 | ||
|
|
cea8a8dfd9 | ||
|
|
931ddb5fc0 | ||
|
|
53d5932a63 | ||
|
|
adc0882b53 | ||
|
|
daeb35986e | ||
|
|
75a70e31c6 | ||
|
|
f869a657d3 | ||
|
|
fcbb01480b | ||
|
|
655634c236 | ||
|
|
b21fbe8b7e | ||
|
|
21bb972b85 | ||
|
|
0fcf2fb285 | ||
|
|
9ce448e0a4 | ||
|
|
ca8cb4480a | ||
|
|
5e4bdf4a2c | ||
|
|
011422636f | ||
|
|
dace15a300 | ||
|
|
f3134943a4 | ||
|
|
5eb0a7e114 | ||
|
|
48ead2727f | ||
|
|
dbb226a7a4 | ||
|
|
04352e92cc | ||
|
|
23975591bc | ||
|
|
9027adebc2 | ||
|
|
ce05c8af80 | ||
|
|
cfb870a323 | ||
|
|
22709379dd | ||
|
|
4aafcfb40f | ||
|
|
d53aa0c816 | ||
|
|
21976bf94f | ||
|
|
434e8ac8fc | ||
|
|
2376532e3d | ||
|
|
8b2fbe3f34 | ||
|
|
f991ae44ab | ||
|
|
7eb8b76d19 | ||
|
|
7da9c0b644 | ||
|
|
46a75498d2 | ||
|
|
c2208eb454 | ||
|
|
f9c43d4a8a | ||
|
|
32ae5b4af0 | ||
|
|
2345f89625 | ||
|
|
46a80b731b | ||
|
|
2781808a96 | ||
|
|
5af1bc523d | ||
|
|
9742d29e51 | ||
|
|
30fda48397 | ||
|
|
6888728be5 | ||
|
|
8c7fbc6210 | ||
|
|
4c4e224d31 | ||
|
|
ac98f72005 | ||
|
|
ddebceb70c | ||
|
|
d8e5c3b461 | ||
|
|
0bfe70afcd | ||
|
|
7488a5dc27 | ||
|
|
4914987998 | ||
|
|
abd30065d0 | ||
|
|
68960f1221 | ||
|
|
836d1ebbd1 | ||
|
|
f96c830a66 | ||
|
|
ecd1fc28d2 | ||
|
|
da0874682f | ||
|
|
e851c1ad99 | ||
|
|
93fd7088ba | ||
|
|
cab440a05b | ||
|
|
f03f88a0f7 | ||
|
|
841fbf9f53 | ||
|
|
bed0c09ad9 | ||
|
|
2497476009 | ||
|
|
5e2841384f | ||
|
|
49de587b2c | ||
|
|
feea47a206 | ||
|
|
f4b1b277cf | ||
|
|
e4608c983b | ||
|
|
f147b50332 | ||
|
|
b1b16f718b | ||
|
|
871eb25dc3 | ||
|
|
793515bac2 | ||
|
|
4496842a86 | ||
|
|
0687238a97 | ||
|
|
cd9120bc15 | ||
|
|
77388979d7 | ||
|
|
d9198306a8 | ||
|
|
093e32658b | ||
|
|
e6821c94c9 | ||
|
|
2fbf8c4379 | ||
|
|
257c478ef3 | ||
|
|
9ce7e61434 | ||
|
|
42fa7ac1a3 | ||
|
|
85d43c76ca | ||
|
|
9f4d837e54 | ||
|
|
27db486275 | ||
|
|
2df60d7862 | ||
|
|
7850587517 | ||
|
|
b924278b03 | ||
|
|
adb16ca8b4 | ||
|
|
42383514ff | ||
|
|
2609db529e | ||
|
|
632385c6bc | ||
|
|
4c4519f679 | ||
|
|
98cded3a79 | ||
|
|
8c37e3995b | ||
|
|
1281bce438 | ||
|
|
aaf0fd2ea8 | ||
|
|
8e249b61d9 | ||
|
|
250418bab9 | ||
|
|
70ec8b4ae2 | ||
|
|
e48f3f05be | ||
|
|
a60c0376c4 | ||
|
|
4bb5ffa1ec | ||
|
|
215491b24c | ||
|
|
e90c6925cb | ||
|
|
8967e7301b | ||
|
|
0977d875ea | ||
|
|
2fec95e8c9 | ||
|
|
7665ae6d96 | ||
|
|
b58f02d5bf | ||
|
|
e09d6f40e5 | ||
|
|
c45f6a4f4d | ||
|
|
ffaa5c114e | ||
|
|
62d8bbddba | ||
|
|
fe8ab55890 | ||
|
|
68e31b5407 | ||
|
|
1640f568ea | ||
|
|
15f058d9a2 | ||
|
|
e9c222737d | ||
|
|
4b48806393 | ||
|
|
2a2ab113c3 | ||
|
|
db0f633415 | ||
|
|
b7f04de30d | ||
|
|
a89eadf7e8 | ||
|
|
d10db455ea | ||
|
|
1605c6884c | ||
|
|
b2661f9f69 | ||
|
|
8ad841b0f2 | ||
|
|
3202f7602b | ||
|
|
6f8d986dc3 | ||
|
|
b5ce88df10 | ||
|
|
abdf3d73d6 | ||
|
|
0255d172f5 | ||
|
|
8cebcbb984 | ||
|
|
8eb680d0e9 | ||
|
|
73c312e81a | ||
|
|
0ba19214e1 | ||
|
|
0568dc979a | ||
|
|
0fda02c859 | ||
|
|
8411f656d0 | ||
|
|
cc4a6675f9 | ||
|
|
4e4bc7abe0 | ||
|
|
95d06a3995 | ||
|
|
04b4d0ccd1 | ||
|
|
f3c440359b | ||
|
|
bacf337b4e | ||
|
|
f5f5bc6a3c | ||
|
|
da1a5fd6be | ||
|
|
fb6f778c6e | ||
|
|
c4b3f39bfe | ||
|
|
2b76b2618f | ||
|
|
c77125131c | ||
|
|
a88aea1c2c | ||
|
|
5066381dcc | ||
|
|
5c54f6c5e9 | ||
|
|
dc59488638 | ||
|
|
8ea6932dc2 | ||
|
|
617d68a894 | ||
|
|
5845b12fba | ||
|
|
1c87625014 | ||
|
|
2ba399778b | ||
|
|
ddff8605c0 | ||
|
|
15ab7be0d8 | ||
|
|
55f8adc328 | ||
|
|
2ef0652cca | ||
|
|
57f3cc3094 | ||
|
|
333ad532cb | ||
|
|
1d55f3bf03 | ||
|
|
5d3a71bd66 | ||
|
|
d7918ad939 | ||
|
|
cd468bc7a6 | ||
|
|
35f6daa68a | ||
|
|
a4d9e8c38e | ||
|
|
2e5c8082ab | ||
|
|
f36b154181 |
@@ -238,7 +238,7 @@ def _get_notebook_python_version(notebook_path: str) -> str:
|
||||
|
||||
# Look for the python version specification pattern
|
||||
re_match = re.search(
|
||||
"python version = (\d+\.\d+)", markdown, flags=re.IGNORECASE
|
||||
r"python version = (\d+\.\d+)", markdown, flags=re.IGNORECASE
|
||||
)
|
||||
if re_match:
|
||||
# get the version number
|
||||
@@ -365,7 +365,7 @@ def process_and_execute_notebook(
|
||||
# Use gcloud to get tail
|
||||
try:
|
||||
result.error_message = subprocess.check_output(
|
||||
["gsutil", "cat", "-r", "-1000", log_file_uri], encoding="UTF-8"
|
||||
["gcloud", "storage", "cat", "--range", "-1000", log_file_uri], encoding="UTF-8"
|
||||
)
|
||||
except Exception as error:
|
||||
result.error_message = str(error)
|
||||
|
||||
@@ -56,8 +56,8 @@ def execute_notebook(
|
||||
print("\n=== DOWNLOAD EXECUTED NOTEBOOK ===\n")
|
||||
print(f"Please debug the executed notebook by downloading the executed notebook:")
|
||||
|
||||
print("Option 1. Using gsutil. Run the following command in your terminal.")
|
||||
print(f'\tgsutil cp "{output_file_or_uri}" .')
|
||||
print("Option 1. Using gcloud storage. Run the following command in your terminal.")
|
||||
print(f'\tgcloud storage cp "{output_file_or_uri}" .')
|
||||
|
||||
print("Option 2. Using this link.")
|
||||
print(f"\thttps://storage.googleapis.com/{output_file_or_uri[5:]}")
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
notebooks/official/vizier/gapic-vizier-multi-objective-optimization.ipynb
|
||||
notebooks/official/pipelines/lightweight_functions_component_io_kfp.ipynb
|
||||
notebooks/official/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb
|
||||
notebooks/official/custom/custom-tabular-bq-managed-dataset.ipynb
|
||||
.cloud-build/tests/python_version_test.ipynb
|
||||
|
||||
@@ -108,7 +108,7 @@ class VertexAIInstallProprocessor(Preprocessor):
|
||||
if "google-cloud-aiplatform" not in content:
|
||||
return content
|
||||
return (
|
||||
f"gsutil cp {self.vertex_ai_wheel} google-cloud-aiplatform.whl\n" +
|
||||
f"gcloud storage cp {self.vertex_ai_wheel} google-cloud-aiplatform.whl\n" +
|
||||
content.replace("google-cloud-aiplatform\n", "google-cloud-aiplatform.whl\n")
|
||||
.replace("google-cloud-aiplatform ", "google-cloud-aiplatform.whl ")
|
||||
)
|
||||
|
||||
@@ -15,7 +15,7 @@ def download_file(bucket_name: str, blob_name: str, destination_file: str) -> st
|
||||
remote_file_path = "".join(["gs://", "/".join([bucket_name, blob_name])])
|
||||
|
||||
subprocess.check_output(
|
||||
["gsutil", "cp", remote_file_path, destination_file], encoding="UTF-8"
|
||||
["gcloud", "storage", "cp", remote_file_path, destination_file], encoding="UTF-8"
|
||||
)
|
||||
|
||||
return destination_file
|
||||
@@ -27,7 +27,7 @@ def upload_file(
|
||||
) -> str:
|
||||
"""Copies a local file to a GCS path"""
|
||||
subprocess.check_output(
|
||||
["gsutil", "cp", local_file_path, remote_file_path], encoding="UTF-8"
|
||||
["gcloud", "storage", "cp", local_file_path, remote_file_path], encoding="UTF-8"
|
||||
)
|
||||
|
||||
return remote_file_path
|
||||
|
||||
@@ -7,11 +7,11 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: '3.x'
|
||||
python-version: '3.14'
|
||||
- name: Fetch pull request branch
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Fetch base main branch
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
# 2. To lint specific notebooks:
|
||||
# docker run -v ${PWD}:/setup/app gcr.io/python-docs-samples-tests/notebook_linter:latest notebooks/1.ipynb notebooks/2.ipynb
|
||||
|
||||
FROM python:3.13
|
||||
FROM python:3.14
|
||||
|
||||
WORKDIR setup
|
||||
|
||||
|
||||
@@ -2,9 +2,9 @@ git+https://github.com/tensorflow/docs
|
||||
ipython
|
||||
jupyter
|
||||
nbconvert
|
||||
black==25.1.0
|
||||
pyupgrade==3.19.1
|
||||
isort==6.0.1
|
||||
flake8==7.1.1
|
||||
black==25.12.0
|
||||
pyupgrade==3.21.0
|
||||
isort==7.0.0
|
||||
flake8==7.3.0
|
||||
nbqa==1.9.1
|
||||
|
||||
|
||||
@@ -29,4 +29,5 @@
|
||||
/vertex_model_garden/model_oss/vllm @kathyyu-google
|
||||
/vertex_model_garden/benchmarking_reports @lavraicse
|
||||
/vertex_model_garden/model_oss/autogluon @lavraicse
|
||||
/vertex_distributed_training/a3mega/llama-3-8b-nemo-pretraining @mstyer-google @erwinh85 @mchrestkha
|
||||
|
||||
|
||||
+1
-1
@@ -148,7 +148,7 @@ implementation:
|
||||
|
||||
# Downloading the model archive from GCS
|
||||
# TODO: Fix gsutil bugs (requires project ID, has auth issues) and use gsutil instead.
|
||||
# gsutil cp "$model_archive_uri" "$model_archive_local_path"
|
||||
# gcloud storage cp "$model_archive_uri" "$model_archive_local_path"
|
||||
pip install google-cloud-storage
|
||||
python -c '
|
||||
import sys
|
||||
|
||||
@@ -24,12 +24,12 @@ implementation:
|
||||
|
||||
# Checking whether the URI points to a single blob, a directory or a URI pattern
|
||||
# URI points to a blob when that URI does not end with slash and listing that URI only yields the same URI
|
||||
if [[ "$uri" != */ ]] && (gsutil ls "$uri" | grep --fixed-strings --line-regexp "$uri"); then
|
||||
if [[ "$uri" != */ ]] && (gcloud storage ls "$uri" | grep --fixed-strings --line-regexp "$uri"); then
|
||||
mkdir -p "$(dirname "$output_path")"
|
||||
gsutil -m cp -r "$uri" "$output_path"
|
||||
gcloud storage cp --recursive "$uri" "$output_path"
|
||||
else
|
||||
mkdir -p "$output_path" # When source path is a directory, gsutil requires the destination to also be a directory
|
||||
gsutil -m rsync -r "$uri" "$output_path" # gsutil cp has different path handling than Linux cp. It always puts the source directory (name) inside the destination directory. gsutil rsync does not have that problem.
|
||||
gcloud storage rsync --recursive "$uri" "$output_path" # gsutil cp has different path handling than Linux cp. It always puts the source directory (name) inside the destination directory. gsutil rsync does not have that problem.
|
||||
fi
|
||||
- inputValue: GCS path
|
||||
- outputPath: Data
|
||||
|
||||
+1492
-1505
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1,3 +1,3 @@
|
||||
torch==2.2.0
|
||||
torch==2.8.0
|
||||
torchvision==0.9.1
|
||||
tensorboard==2.5.0
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
torch==2.2.0
|
||||
torch==2.7.0
|
||||
torchvision==0.9.1
|
||||
tensorboard==2.5.0
|
||||
+3
-3
@@ -110,7 +110,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $gcs_output_uri_prefix"
|
||||
"! gcloud storage ls $gcs_output_uri_prefix"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -192,7 +192,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil cp -r $gcs_output_uri_prefix/model ./model_server/"
|
||||
"! gcloud storage cp --recursive $gcs_output_uri_prefix/model ./model_server/"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -556,7 +556,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil rm -rf $gcs_output_uri_prefix"
|
||||
"! gcloud storage rm --recursive --continue-on-error $gcs_output_uri_prefix"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
+1
-1
@@ -412,7 +412,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $gcs_output_uri_prefix"
|
||||
"! gcloud storage ls $gcs_output_uri_prefix"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+1
-1
@@ -77,4 +77,4 @@ echo "After the job is completed successfully, model files will be saved at $JOB
|
||||
|
||||
# # Verify the model was exported
|
||||
# echo "Verify the model was exported:"
|
||||
# gsutil ls ${JOB_DIR}/
|
||||
# gcloud storage ls ${JOB_DIR}/
|
||||
|
||||
+1
-1
@@ -34,4 +34,4 @@ RUN echo "service_envelope=json\n" "inference_address=http://0.0.0.0:${AIP_H
|
||||
USER model-server
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["echo", "AIP_STORAGE_URI=${AIP_STORAGE_URI}", ";", "gsutil", "cp", "-r", "${AIP_STORAGE_URI}/${MODEL_NAME}.mar", "/home/model-server/model-store/", ";", "ls", "-ltr", "/home/model-server/model-store/", ";", "torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "${MODEL_NAME}=${MODEL_NAME}.mar", "--model-store", "/home/model-server/model-store"]
|
||||
CMD ["echo", "AIP_STORAGE_URI=${AIP_STORAGE_URI}", ";", "gcloud", "storage", "cp", "--recursive", "${AIP_STORAGE_URI}/${MODEL_NAME}.mar", "/home/model-server/model-store/", ";", "ls", "-ltr", "/home/model-server/model-store/", ";", "torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "${MODEL_NAME}=${MODEL_NAME}.mar", "--model-store", "/home/model-server/model-store"]
|
||||
|
||||
+1
-1
@@ -67,4 +67,4 @@ echo "After the job is completed successfully, model files will be saved at $JOB
|
||||
|
||||
# # Verify the model was exported
|
||||
# echo "Verify the model was exported:"
|
||||
# gsutil ls ${JOB_DIR}/
|
||||
# gcloud storage ls ${JOB_DIR}/
|
||||
|
||||
+4
-9
@@ -478,8 +478,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -498,8 +497,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -582,8 +580,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download the sample data into your RAW_DATA_PATH\n",
|
||||
"! gsutil cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $RAW_DATA_PATH"
|
||||
]
|
||||
"! gcloud storage cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $RAW_DATA_PATH" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
@@ -1621,9 +1618,7 @@
|
||||
"! gcloud scheduler jobs delete $SIMULATOR_SCHEDULER_JOB --quiet\n",
|
||||
"\n",
|
||||
"# Delete Cloud Storage objects that were created.\n",
|
||||
"! gsutil -m rm -r $PIPELINE_ROOT\n",
|
||||
"! gsutil -m rm -r $TRAINING_ARTIFACTS_DIR"
|
||||
]
|
||||
"! gcloud storage rm --recursive $PIPELINE_ROOT\n", "! gcloud storage rm --recursive $TRAINING_ARTIFACTS_DIR" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+75
-44
@@ -398,6 +398,7 @@
|
||||
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
@@ -472,7 +473,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -492,7 +493,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -565,7 +566,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copy the sample data into your DATA_PATH\n",
|
||||
"! gsutil cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $DATA_PATH"
|
||||
"! gcloud storage cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $DATA_PATH"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -579,11 +580,15 @@
|
||||
"# Set hyperparameters.\n",
|
||||
"BATCH_SIZE = 8 # @param {type:\"integer\"} Training and prediction batch size.\n",
|
||||
"TRAINING_LOOPS = 5 # @param {type:\"integer\"} Number of training iterations.\n",
|
||||
"STEPS_PER_LOOP = 2 # @param {type:\"integer\"} Number of driver steps per training iteration.\n",
|
||||
"STEPS_PER_LOOP = (\n",
|
||||
" 2 # @param {type:\"integer\"} Number of driver steps per training iteration.\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Set MovieLens simulation environment parameters.\n",
|
||||
"RANK_K = 20 # @param {type:\"integer\"} Rank for matrix factorization in the MovieLens environment; also the observation dimension.\n",
|
||||
"NUM_ACTIONS = 20 # @param {type:\"integer\"} Number of actions (movie items) to choose from.\n",
|
||||
"NUM_ACTIONS = (\n",
|
||||
" 20 # @param {type:\"integer\"} Number of actions (movie items) to choose from.\n",
|
||||
")\n",
|
||||
"PER_ARM = False # Use the non-per-arm version of the MovieLens environment.\n",
|
||||
"\n",
|
||||
"# Set agent parameters.\n",
|
||||
@@ -621,7 +626,8 @@
|
||||
"source": [
|
||||
"# Define RL environment.\n",
|
||||
"env = movielens_py_environment.MovieLensPyEnvironment(\n",
|
||||
" DATA_PATH, RANK_K, BATCH_SIZE, num_movies=NUM_ACTIONS, csv_delimiter=\"\\t\")\n",
|
||||
" DATA_PATH, RANK_K, BATCH_SIZE, num_movies=NUM_ACTIONS, csv_delimiter=\"\\t\"\n",
|
||||
")\n",
|
||||
"environment = tf_py_environment.TFPyEnvironment(env)\n",
|
||||
"\n",
|
||||
"# Define RL agent/algorithm.\n",
|
||||
@@ -631,7 +637,8 @@
|
||||
" tikhonov_weight=TIKHONOV_WEIGHT,\n",
|
||||
" alpha=AGENT_ALPHA,\n",
|
||||
" dtype=tf.float32,\n",
|
||||
" accepts_per_arm_features=PER_ARM)\n",
|
||||
" accepts_per_arm_features=PER_ARM,\n",
|
||||
")\n",
|
||||
"print(\"TimeStep Spec (for each batch):\\n\", agent.time_step_spec, \"\\n\")\n",
|
||||
"print(\"Action Spec (for each batch):\\n\", agent.action_spec, \"\\n\")\n",
|
||||
"print(\"Reward Spec (for each batch):\\n\", environment.reward_spec(), \"\\n\")\n",
|
||||
@@ -639,7 +646,8 @@
|
||||
"# Define RL metric.\n",
|
||||
"optimal_reward_fn = functools.partial(\n",
|
||||
" environment_utilities.compute_optimal_reward_with_movielens_environment,\n",
|
||||
" environment=environment)\n",
|
||||
" environment=environment,\n",
|
||||
")\n",
|
||||
"regret_metric = tf_bandit_metrics.RegretMetric(optimal_reward_fn)\n",
|
||||
"metrics = [regret_metric]"
|
||||
]
|
||||
@@ -704,35 +712,38 @@
|
||||
" if training_data_spec_transformation_fn is None:\n",
|
||||
" data_spec = agent.policy.trajectory_spec\n",
|
||||
" else:\n",
|
||||
" data_spec = training_data_spec_transformation_fn(\n",
|
||||
" agent.policy.trajectory_spec)\n",
|
||||
" replay_buffer = trainer.get_replay_buffer(data_spec, environment.batch_size,\n",
|
||||
" steps_per_loop)\n",
|
||||
" data_spec = training_data_spec_transformation_fn(agent.policy.trajectory_spec)\n",
|
||||
" replay_buffer = trainer.get_replay_buffer(\n",
|
||||
" data_spec, environment.batch_size, steps_per_loop\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # `step_metric` records the number of individual rounds of bandit interaction;\n",
|
||||
" # that is, (number of trajectories) * batch_size.\n",
|
||||
" step_metric = tf_metrics.EnvironmentSteps()\n",
|
||||
" metrics = [\n",
|
||||
" tf_metrics.NumberOfEpisodes(),\n",
|
||||
" tf_metrics.AverageEpisodeLengthMetric(batch_size=environment.batch_size)\n",
|
||||
" tf_metrics.AverageEpisodeLengthMetric(batch_size=environment.batch_size),\n",
|
||||
" ]\n",
|
||||
" if additional_metrics:\n",
|
||||
" metrics += additional_metrics\n",
|
||||
"\n",
|
||||
" if isinstance(environment.reward_spec(), dict):\n",
|
||||
" metrics += [tf_metrics.AverageReturnMultiMetric(\n",
|
||||
" reward_spec=environment.reward_spec(),\n",
|
||||
" batch_size=environment.batch_size)]\n",
|
||||
" else:\n",
|
||||
" metrics += [\n",
|
||||
" tf_metrics.AverageReturnMetric(batch_size=environment.batch_size)]\n",
|
||||
" tf_metrics.AverageReturnMultiMetric(\n",
|
||||
" reward_spec=environment.reward_spec(), batch_size=environment.batch_size\n",
|
||||
" )\n",
|
||||
" ]\n",
|
||||
" else:\n",
|
||||
" metrics += [tf_metrics.AverageReturnMetric(batch_size=environment.batch_size)]\n",
|
||||
"\n",
|
||||
" # Store intermediate metric results, indexed by metric names.\n",
|
||||
" metric_results = defaultdict(list)\n",
|
||||
"\n",
|
||||
" if training_data_spec_transformation_fn is not None:\n",
|
||||
" def add_batch_fn(data): return replay_buffer.add_batch(training_data_spec_transformation_fn(data)) \n",
|
||||
" \n",
|
||||
"\n",
|
||||
" def add_batch_fn(data):\n",
|
||||
" return replay_buffer.add_batch(training_data_spec_transformation_fn(data))\n",
|
||||
"\n",
|
||||
" else:\n",
|
||||
" add_batch_fn = replay_buffer.add_batch\n",
|
||||
"\n",
|
||||
@@ -742,10 +753,12 @@
|
||||
" env=environment,\n",
|
||||
" policy=agent.collect_policy,\n",
|
||||
" num_steps=steps_per_loop * environment.batch_size,\n",
|
||||
" observers=observers)\n",
|
||||
" observers=observers,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" training_loop = trainer.get_training_loop_fn(\n",
|
||||
" driver, replay_buffer, agent, steps_per_loop)\n",
|
||||
" driver, replay_buffer, agent, steps_per_loop\n",
|
||||
" )\n",
|
||||
" saver = policy_saver.PolicySaver(agent.policy)\n",
|
||||
"\n",
|
||||
" for _ in range(training_loops):\n",
|
||||
@@ -783,7 +796,8 @@
|
||||
" environment=environment,\n",
|
||||
" training_loops=TRAINING_LOOPS,\n",
|
||||
" steps_per_loop=STEPS_PER_LOOP,\n",
|
||||
" additional_metrics=metrics)\n",
|
||||
" additional_metrics=metrics,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"tf.profiler.experimental.stop()"
|
||||
]
|
||||
@@ -1092,11 +1106,15 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"RUN_HYPERPARAMETER_TUNING = True # Execute hyperparameter tuning instead of regular training.\n",
|
||||
"RUN_HYPERPARAMETER_TUNING = (\n",
|
||||
" True # Execute hyperparameter tuning instead of regular training.\n",
|
||||
")\n",
|
||||
"TRAIN_WITH_BEST_HYPERPARAMETERS = False # Do not train.\n",
|
||||
"\n",
|
||||
"HPTUNING_RESULT_DIR = \"hptuning/\" # @param {type: \"string\"} Directory to store the best hyperparameter(s) in `BUCKET_NAME` and locally (temporarily).\n",
|
||||
"HPTUNING_RESULT_PATH = os.path.join(HPTUNING_RESULT_DIR, \"result.json\") # @param {type: \"string\"} Path to the file containing the best hyperparameter(s)."
|
||||
"HPTUNING_RESULT_PATH = os.path.join(\n",
|
||||
" HPTUNING_RESULT_DIR, \"result.json\"\n",
|
||||
") # @param {type: \"string\"} Path to the file containing the best hyperparameter(s)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1124,7 +1142,7 @@
|
||||
" image_uri: str,\n",
|
||||
" args: List[str],\n",
|
||||
" location: str = \"us-central1\",\n",
|
||||
" api_endpoint: str = \"us-central1-aiplatform.googleapis.com\"\n",
|
||||
" api_endpoint: str = \"us-central1-aiplatform.googleapis.com\",\n",
|
||||
") -> None:\n",
|
||||
" \"\"\"Creates a hyperparameter tuning job using a custom container.\n",
|
||||
"\n",
|
||||
@@ -1197,8 +1215,8 @@
|
||||
"\n",
|
||||
" # Create job\n",
|
||||
" response = client.create_hyperparameter_tuning_job(\n",
|
||||
" parent=parent,\n",
|
||||
" hyperparameter_tuning_job=hyperparameter_tuning_job)\n",
|
||||
" parent=parent, hyperparameter_tuning_job=hyperparameter_tuning_job\n",
|
||||
" )\n",
|
||||
" job_id = response.name.split(\"/\")[-1]\n",
|
||||
" print(\"Job ID:\", job_id)\n",
|
||||
" print(\"Job config:\", response)\n",
|
||||
@@ -1242,7 +1260,8 @@
|
||||
" image_uri=f\"gcr.io/{PROJECT_ID}/{HPTUNING_TRAINING_CONTAINER}:latest\",\n",
|
||||
" args=args,\n",
|
||||
" location=REGION,\n",
|
||||
" api_endpoint=f\"{REGION}-aiplatform.googleapis.com\")"
|
||||
" api_endpoint=f\"{REGION}-aiplatform.googleapis.com\",\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1292,7 +1311,8 @@
|
||||
" name = client.hyperparameter_tuning_job_path(\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" hyperparameter_tuning_job=hyperparameter_tuning_job_id)\n",
|
||||
" hyperparameter_tuning_job=hyperparameter_tuning_job_id,\n",
|
||||
" )\n",
|
||||
" response = client.get_hyperparameter_tuning_job(name=name)\n",
|
||||
" return response"
|
||||
]
|
||||
@@ -1313,7 +1333,8 @@
|
||||
" location=REGION,\n",
|
||||
" api_endpoint=f\"{REGION}-aiplatform.googleapis.com\")\n",
|
||||
" if response.state.name == 'JOB_STATE_SUCCEEDED':\n",
|
||||
" print(\"Job succeeded.\\nJob Time:\", response.update_time - response.create_time)\n",
|
||||
" print(\"Job succeeded.\n",
|
||||
"Job Time:\", response.update_time - response.create_time)\n",
|
||||
" trials = response.trials\n",
|
||||
" print(\"Trials:\", trials)\n",
|
||||
" break\n",
|
||||
@@ -1348,8 +1369,8 @@
|
||||
"if trials:\n",
|
||||
" # Dict mapping from metric names to the best metric values seen so far\n",
|
||||
" best_objective_values = dict.fromkeys(\n",
|
||||
" [metric.metric_id for metric in trials[0].final_measurement.metrics],\n",
|
||||
" -np.inf)\n",
|
||||
" [metric.metric_id for metric in trials[0].final_measurement.metrics], -np.inf\n",
|
||||
" )\n",
|
||||
" # Dict mapping from metric names to a list of the best combination(s) of\n",
|
||||
" # hyperparameter(s). Each combination is a dict mapping from hyperparameter\n",
|
||||
" # names to their values.\n",
|
||||
@@ -1358,12 +1379,13 @@
|
||||
" # `final_measurement` and `parameters` are `RepeatedComposite` objects.\n",
|
||||
" # Reference the structure above to extract the value of your interest.\n",
|
||||
" for metric in trial.final_measurement.metrics:\n",
|
||||
" params = {\n",
|
||||
" param.parameter_id: param.value for param in trial.parameters}\n",
|
||||
" params = {param.parameter_id: param.value for param in trial.parameters}\n",
|
||||
" if metric.value > best_objective_values[metric.metric_id]:\n",
|
||||
" best_params[metric.metric_id] = [params]\n",
|
||||
" elif metric.value == best_objective_values[metric.metric_id]:\n",
|
||||
" best_params[param.parameter_id].append(params) # Handle cases where multiple hyperparameter values lead to the same performance.\n",
|
||||
" best_params[param.parameter_id].append(\n",
|
||||
" params\n",
|
||||
" ) # Handle cases where multiple hyperparameter values lead to the same performance.\n",
|
||||
" print(\"Best hyperparameter value(s):\")\n",
|
||||
" for metric, params in best_params.items():\n",
|
||||
" print(f\"Metric={metric}: {sorted(params)}\")\n",
|
||||
@@ -1443,7 +1465,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PREDICTION_CONTAINER = \"prediction-custom-container\" # @param {type:\"string\"} Name of the container image."
|
||||
"PREDICTION_CONTAINER = (\n",
|
||||
" \"prediction-custom-container\" # @param {type:\"string\"} Name of the container image.\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1475,7 +1499,7 @@
|
||||
" machineType: 'E2_HIGHCPU_8'\"\"\".format(\n",
|
||||
" PROJECT_ID=PROJECT_ID,\n",
|
||||
" PREDICTION_CONTAINER=PREDICTION_CONTAINER,\n",
|
||||
" ARTIFACTS_DIR=ARTIFACTS_DIR\n",
|
||||
" ARTIFACTS_DIR=ARTIFACTS_DIR,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"with open(\"cloudbuild.yaml\", \"w\") as fp:\n",
|
||||
@@ -1592,8 +1616,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"RUN_HYPERPARAMETER_TUNING = False # Execute regular training instead of hyperparameter tuning.\n",
|
||||
"TRAIN_WITH_BEST_HYPERPARAMETERS = True # @param {type:\"bool\"} Whether to use learned hyperparameters in training."
|
||||
"RUN_HYPERPARAMETER_TUNING = (\n",
|
||||
" False # Execute regular training instead of hyperparameter tuning.\n",
|
||||
")\n",
|
||||
"TRAIN_WITH_BEST_HYPERPARAMETERS = (\n",
|
||||
" True # @param {type:\"bool\"} Whether to use learned hyperparameters in training.\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1633,10 +1661,12 @@
|
||||
"job = aiplatform.CustomContainerTrainingJob(\n",
|
||||
" display_name=\"train-movielens\",\n",
|
||||
" container_uri=f\"gcr.io/{PROJECT_ID}/{HPTUNING_TRAINING_CONTAINER}:latest\",\n",
|
||||
" command=[\"python3\", \"-m\", \"src.training.task\"] + args, # Pass in training arguments, including hyperparameters.\n",
|
||||
" command=[\"python3\", \"-m\", \"src.training.task\"]\n",
|
||||
" + args, # Pass in training arguments, including hyperparameters.\n",
|
||||
" model_serving_container_image_uri=f\"gcr.io/{PROJECT_ID}/{PREDICTION_CONTAINER}:latest\",\n",
|
||||
" model_serving_container_predict_route=\"/predict\",\n",
|
||||
" model_serving_container_health_route=\"/health\")\n",
|
||||
" model_serving_container_health_route=\"/health\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"Training Spec:\", job._managed_model)\n",
|
||||
"\n",
|
||||
@@ -1645,7 +1675,8 @@
|
||||
" replica_count=1,\n",
|
||||
" machine_type=\"n1-standard-4\",\n",
|
||||
" accelerator_type=\"ACCELERATOR_TYPE_UNSPECIFIED\",\n",
|
||||
" accelerator_count=0)"
|
||||
" accelerator_count=0,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1784,7 +1815,7 @@
|
||||
"! gcloud ai models delete $model.name --quiet\n",
|
||||
"\n",
|
||||
"# Delete Cloud Storage objects that were created\n",
|
||||
"! gsutil -m rm -r $ARTIFACTS_DIR"
|
||||
"! gcloud storage rm --recursive $ARTIFACTS_DIR"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+2
-2
@@ -324,7 +324,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $gcs_output_uri_prefix"
|
||||
"! gcloud storage ls $gcs_output_uri_prefix"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -344,7 +344,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil rm -rf $gcs_output_uri_prefix"
|
||||
"! gcloud storage rm --recursive --continue-on-error $gcs_output_uri_prefix"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+2
-2
@@ -328,7 +328,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $gcs_output_uri_prefix"
|
||||
"! gcloud storage ls $gcs_output_uri_prefix"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -348,7 +348,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil rm -rf $gcs_output_uri_prefix"
|
||||
"! gcloud storage rm --recursive --continue-on-error $gcs_output_uri_prefix"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+2
-2
@@ -341,7 +341,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $gcs_output_uri_prefix"
|
||||
"! gcloud storage ls $gcs_output_uri_prefix"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -361,7 +361,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil rm -rf $gcs_output_uri_prefix"
|
||||
"! gcloud storage rm --recursive --continue-on-error $gcs_output_uri_prefix"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+126
@@ -0,0 +1,126 @@
|
||||
# Vertex AI Training: Llama 3.1 8B pre-training using Nvidia A3 Mega VMs (H100)
|
||||
This document provides a step-by-step guide for pre-training a Llama 3.1 8B model on the `en-wiki` dataset using multiple [Vertex AI Custom Training](https://cloud.google.com/vertex-ai/docs/training/overview) `a3-megagpu-8g` nodes.
|
||||
|
||||
We will use a custom container based on NVIDIA's [NeMo Framework](https://docs.nvidia.com/nemo-framework/user-guide/24.07/overview.html) to demonstrate a scalable, multi-node training workflow. All required artifacts and commands are included.
|
||||
|
||||
## 1. Prerequisites
|
||||
|
||||
### 1.1. Google Cloud Project setup
|
||||
- **Enable APIs:** Ensure the Vertex AI API is [enabled for your project](http://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com).
|
||||
- **H100 Mega Quota:** A3 Mega VMs are powered by H100 GPUs. Request quota for `custom_model_training_nvidia_h100_mega_gpus` in one of the [supported regions](https://cloud.google.com/vertex-ai/docs/general/locations#accelerator_support). If using Spot VMs, request `custom_model_training_preemptible_nvidia_h100_mega_gpus` quota instead.
|
||||
- **Reservations (Optional but recommended):** For guaranteed capacity, [create a reservation](https://cloud.google.com/compute/docs/instances/reservations-shared) and ensure the reservation is shared with the Vertex AI service account. This guide requires a minimum of **16 H100 GPUs** (2 full A3 Mega nodes).
|
||||
|
||||
### 1.2. GCS bucket
|
||||
Create a [Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) in the same region where you have quota. If you're using Hierarchical Namespace for your bucket, you may need to update permissions of the Vertex AI Custom Code Service Agent .
|
||||
|
||||
This bucket is used for:
|
||||
- Staging the training application.
|
||||
- Storing model checkpoints and logs.
|
||||
- Storing data if you use your own data.
|
||||
|
||||
|
||||
## 2. Setup & configuration
|
||||
|
||||
### 2.1. Clone the repo
|
||||
First clone the repo into your development environment.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
|
||||
```
|
||||
|
||||
Navigate to the root folder for this sample.
|
||||
|
||||
### 2.2. Environment Setup
|
||||
First, configure your local environment. These variables are used in subsequent commands.
|
||||
|
||||
```bash
|
||||
# Required: Update with your values
|
||||
export PROJECT_ID="<your-project-id>"
|
||||
export REPOSITORY="<your-artifact-registry-repo-name>" # e.g., "my-containers"
|
||||
export BUCKET="<your-gcs-bucket-name>"
|
||||
|
||||
# Optional: Change if needed
|
||||
export REGION="us-central1"
|
||||
|
||||
# --- Do not change the lines below ---
|
||||
export ARTIFACT_REGISTRY="${REGION}-docker.pkg.dev/${PROJECT_ID}/${REPOSITORY}"
|
||||
export REPO_ROOT=$(git rev-parse --show-toplevel)
|
||||
```
|
||||
|
||||
## 3. Build and push a docker container image to Artifact Registry
|
||||
Normally, you can use any custom training container on Vertex AI Training. In this example you build a NeMo Docker image that is based on the [Nvidia’s NeMo 24.09](https://catalog.ngc.nvidia.com/orgs/nvidia/containers/nemo/tags) image. Use Cloud Build to build and push the container image.
|
||||
|
||||
This document picked NeMo as the demonstrating container since it’s a widely adopted GPU LLM training framework providing high performance and versatile training functionalities.
|
||||
|
||||
In addition to the base image, some customizations are included to form the final prebuilt image:
|
||||
- Some dependencies are installed to integrate with Vertex AI Training.
|
||||
- An entrypoint script that sets up required environments and calls the training job.
|
||||
- Some patches are applied to the NeMo code to let it load the dataset from a GCS bucket.
|
||||
|
||||
Run this command to build the container and push the container into the Google Artifact Registry.
|
||||
|
||||
```bash
|
||||
cd "${REPO_ROOT}/community-content/vertex-distributed-training/a3mega/llama-3-8b-nemo-pretraining"
|
||||
export IMAGE_NAME="vertex-nemo-llama"
|
||||
gcloud builds submit . \
|
||||
--project="${PROJECT_ID}" \
|
||||
--region="${REGION}" \
|
||||
--config=docker/cloudbuild.yml \
|
||||
--substitutions="_ARTIFACT_REGISTRY=${ARTIFACT_REGISTRY},_IMAGE_NAME=${IMAGE_NAME}" \
|
||||
--timeout="2h" \
|
||||
--machine-type="e2-highcpu-32"
|
||||
```
|
||||
|
||||
## 4. Launch the Training Job
|
||||
|
||||
|
||||
### 4.1. Job Configuration File
|
||||
Once the container is built, update the job_config.json to set up the training job.
|
||||
File: job_config.json
|
||||
```json
|
||||
{
|
||||
"project_id": "<project-id>",
|
||||
"region": "<region>",
|
||||
"zone": "<zone if using reservation>",
|
||||
"bucket": "<bucket>",
|
||||
"dataset_bucket": "github-repo/data/third-party/enwiki-latest-pages-articles",
|
||||
"image_uri": "<docker image uri from artifact registry>",
|
||||
"strategy": "spot",
|
||||
"nodes": "2",
|
||||
"machine_type": "a3-megagpu-8g",
|
||||
"gpu_type": "NVIDIA_H100_MEGA_80GB",
|
||||
"gpus_per_node": "8",
|
||||
"recipe_name": "llama3_1_8b_pretrain_a3mega",
|
||||
"job_prefix": "vertex-spot-",
|
||||
"reservation_name": ""
|
||||
}
|
||||
```
|
||||
|
||||
### 4.2 Launch the Training Job
|
||||
|
||||
First, create a Python virtual environment using your tool of choice, then install
|
||||
the requirements specified in `requirements.txt`. Using `pip`, the command would be:
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
Now launch the Vertex AI training job using the provided Python script.
|
||||
|
||||
```bash
|
||||
python3 scripts/launch.py --config_file=job_config.json
|
||||
```
|
||||
|
||||
This script reads job_config.json, defines the cluster specification (2 nodes, 8 GPUs each), and submits the custom training job to Vertex AI.
|
||||
|
||||
## 5. Monitor and Clean Up
|
||||
|
||||
### 5.1. Monitoring
|
||||
Vertex AI Console: Track the job's status in the Google Cloud Console under Vertex AI > Training > Custom Jobs.
|
||||
Logs: View detailed logs in Cloud Logging by filtering for your job name.
|
||||
Checkpoints: Model checkpoints are saved to your GCS bucket at the path specified in your training script's configuration.
|
||||
|
||||
### 5.2. Cleaning Up
|
||||
To avoid ongoing charges, delete the resources you created:
|
||||
- The Artifact Registry image.
|
||||
- The contents of the GCS bucket (checkpoints, logs).
|
||||
- The Vertex AI Custom Job will eventually complete or fail, incurring no further cost.
|
||||
+265
@@ -0,0 +1,265 @@
|
||||
# Reference:
|
||||
# https://github.com/NVIDIA/NeMo-Framework-Launcher/blob/24.07/launcher_scripts/conf/training/llama/llama3_1_8b.yaml
|
||||
name: llama3_1_8b_pretrain_a3mega
|
||||
restore_from_path: null # used when starting from a .nemo file
|
||||
|
||||
trainer:
|
||||
devices: 8
|
||||
num_nodes: 1
|
||||
accelerator: gpu
|
||||
precision: bf16
|
||||
logger: false # logger provided by exp_manager
|
||||
enable_checkpointing: false
|
||||
use_distributed_sampler: false
|
||||
max_epochs: -1 # PTL default. In practice, max_steps will be reached first.
|
||||
max_steps: 30 # consumed_samples = global_step * micro_batch_size * data_parallel_size * accumulate_grad_batches
|
||||
log_every_n_steps: 1
|
||||
val_check_interval: null
|
||||
limit_val_batches: 1
|
||||
limit_test_batches: 1
|
||||
accumulate_grad_batches: 1 # do not modify, grad acc is automatic for training megatron models
|
||||
gradient_clip_val: 1.0
|
||||
benchmark: false
|
||||
enable_model_summary: false # default PTL callback for this does not support model parallelism, instead we log manually
|
||||
|
||||
exp_manager:
|
||||
explicit_log_dir: null
|
||||
exp_dir: /data
|
||||
name: ${name}
|
||||
create_dllogger_logger: true
|
||||
dllogger_logger_kwargs:
|
||||
verbose: true
|
||||
stdout: true
|
||||
json_file: "/data/dllogger.json"
|
||||
create_wandb_logger: false
|
||||
wandb_logger_kwargs:
|
||||
project: null
|
||||
name: null
|
||||
resume_if_exists: true
|
||||
resume_ignore_no_checkpoint: true
|
||||
create_checkpoint_callback: false
|
||||
checkpoint_callback_params:
|
||||
monitor: val_loss
|
||||
save_top_k: 3
|
||||
mode: min
|
||||
always_save_nemo: false # saves nemo file during validation, not implemented for model parallel
|
||||
save_nemo_on_train_end: false # not recommended when training large models on clusters with short time limits
|
||||
filename: 'megatron_gpt--{val_loss:.2f}-{step}-{consumed_samples}'
|
||||
model_parallel_size: ${multiply:${model.tensor_model_parallel_size}, ${model.pipeline_model_parallel_size}}
|
||||
seconds_to_sleep: 5 # Allows node_rank!=0 to sleep and let node0 to init, like preparing data
|
||||
|
||||
model:
|
||||
mcore_gpt: true
|
||||
# specify micro_batch_size, global_batch_size, and model parallelism
|
||||
# gradient accumulation will be done automatically based on data_parallel_size
|
||||
micro_batch_size: 1 # limited by GPU memory
|
||||
global_batch_size: 1024 # will use more micro batches to reach global batch size
|
||||
tensor_model_parallel_size: 1 # intra-layer model parallelism
|
||||
pipeline_model_parallel_size: 2 # inter-layer model parallelism
|
||||
context_parallel_size: 1
|
||||
virtual_pipeline_model_parallel_size: null # interleaved pipeline
|
||||
## Sequence Parallelism
|
||||
# Makes tensor parallelism more memory efficient for LLMs (20B+) by parallelizing layer norms and dropout sequentially
|
||||
# See Reducing Activation Recomputation in Large Transformer Models: https://arxiv.org/abs/2205.05198 for more details.
|
||||
sequence_parallel: false
|
||||
|
||||
fsdp: false
|
||||
fsdp_cpu_offload: true
|
||||
fsdp_sharding_strategy: "full" # Method to shard model states. Available options are 'full', 'hybrid', and 'grad'.
|
||||
fsdp_grad_reduce_dtype: "16" # Gradient reduction data type.
|
||||
fsdp_sharded_checkpoint: false # Store and load FSDP shared checkpoint.
|
||||
fsdp_use_orig_params: false # Set to True to use FSDP for specific peft scheme.
|
||||
|
||||
# Distributed checkpoint setup
|
||||
dist_ckpt_format: "torch_dist" # Set to 'torch_dist' to use PyTorch distributed checkpoint format.
|
||||
dist_ckpt_load_on_device: true # whether to load checkpoint weights directly on GPU or to CPU
|
||||
dist_ckpt_parallel_save: true # if true, each worker will write its own part of the dist checkpoint
|
||||
dist_ckpt_parallel_save_within_dp: false # if true, save will be parallelized only within a DP group (whole world otherwise), which might slightly reduce the save overhead
|
||||
dist_ckpt_parallel_load: false # if true, each worker will load part of the dist checkpoint and exchange with NCCL. Might use some extra GPU memory
|
||||
dist_ckpt_torch_dist_multiproc: 2 # number of extra processes per rank used during ckpt save with PyTorch distributed format
|
||||
dist_ckpt_assume_constant_structure: false # set to True only if the state dict structure doesn't change within a single job. Allows caching some computation across checkpoint saves.
|
||||
dist_ckpt_parallel_dist_opt: true # parallel save/load of a DistributedOptimizer. 'True' allows performant save and reshardable checkpoints. Set to 'False' only in order to minimize the number of checkpoint files.
|
||||
dist_ckpt_load_strictness: null # defines checkpoint keys mismatch behavior (only during dist-ckpt load). Choices: assume_ok_unexpected (default - try loading without any check), log_all (log mismatches), raise_all (raise mismatches)
|
||||
|
||||
# model architecture
|
||||
encoder_seq_length: 8192
|
||||
max_position_embeddings: ${.encoder_seq_length}
|
||||
num_layers: 32 # 8b: 32 | 70b: 80 | 405b: 126
|
||||
hidden_size: 4096 # 8b: 4096 | 70b: 8192 | 405b: 16384
|
||||
ffn_hidden_size: 14336 # 8b: 14336 | 70b: 28672 | 405b: 53248
|
||||
num_attention_heads: 32 # 8b: 32 | 70b: 64 | 405b: 128
|
||||
num_query_groups: 8 # Number of query groups for group query attention. If None, normal attention is used. 8b: 8 | 70b: 8 | 405b: 16
|
||||
init_method_std: 0.01 # Standard deviation of the zero mean normal distribution used for weight initialization. 8b: 0.01 | 70b: 0.008944 | 405b: 0.02
|
||||
use_scaled_init_method: true # use scaled residuals initialization
|
||||
hidden_dropout: 0.0 # Dropout probability for hidden state transformer.
|
||||
attention_dropout: 0.0 # Dropout probability for attention
|
||||
ffn_dropout: 0.0 # Dropout probability in the feed-forward layer.
|
||||
kv_channels: null # Projection weights dimension in multi-head attention. Set to hidden_size // num_attention_heads if null
|
||||
apply_query_key_layer_scaling: true # scale Q * K^T by 1 / layer-number.
|
||||
normalization: 'rmsnorm' # Normalization layer to use. Options are 'layernorm', 'rmsnorm'
|
||||
layernorm_epsilon: 1e-5
|
||||
do_layer_norm_weight_decay: false # True means weight decay on all params
|
||||
make_vocab_size_divisible_by: 128 # Pad the vocab size to be divisible by this value for computation efficiency.
|
||||
pre_process: true # add embedding
|
||||
post_process: true # add pooler
|
||||
persist_layer_norm: true # Use of persistent fused layer norm kernel.
|
||||
bias: false # Whether to use bias terms in all weight matrices.
|
||||
activation: 'fast-swiglu' # Options ['gelu', 'geglu', 'swiglu', 'reglu', 'squared-relu', 'fast-geglu', 'fast-swiglu', 'fast-reglu']
|
||||
headscale: false # Whether to learn extra parameters that scale the output of the each self-attention head.
|
||||
transformer_block_type: 'pre_ln' # Options ['pre_ln', 'post_ln', 'normformer']
|
||||
openai_gelu: false # Use OpenAI's GELU instead of the default GeLU
|
||||
normalize_attention_scores: true # Whether to scale the output Q * K^T by 1 / sqrt(hidden_size_per_head). This arg is provided as a configuration option mostly for compatibility with models that have been weight-converted from HF. You almost always want to se this to True.
|
||||
position_embedding_type: 'rope' # Position embedding type. Options ['learned_absolute', 'rope']
|
||||
rotary_percentage: 1.0 # If using position_embedding_type=rope, then the per head dim is multiplied by this.
|
||||
attention_type: 'multihead' # Attention type. Options ['multihead']
|
||||
share_embeddings_and_output_weights: false # Share embedding and output layer weights.
|
||||
scale_positional_embedding: true # This is false for llama3 models. Only used for >= llama3.1.
|
||||
|
||||
# Use GPT2BPETokenizer for test, because the testing dataset is tokenized by this tokenizer.
|
||||
# https://docs.nvidia.com/nemo-framework/user-guide/24.07/playbooks/singlenodepretrain.html#data-download-and-pre-processing
|
||||
tokenizer:
|
||||
library: megatron
|
||||
type: GPT2BPETokenizer
|
||||
model: null # /path/to/tokenizer.model
|
||||
vocab_file: null
|
||||
merge_file: null
|
||||
delimiter: null # only used for tabular tokenizer
|
||||
sentencepiece_legacy: false # Legacy=True allows you to add special tokens to sentencepiece tokenizers.
|
||||
|
||||
# Mixed precision
|
||||
native_amp_init_scale: 4294967296 # 2 ** 32
|
||||
native_amp_growth_interval: 1000
|
||||
hysteresis: 2 # Gradient scale hysteresis
|
||||
fp32_residual_connection: false # Move residual connections to fp32
|
||||
fp16_lm_cross_entropy: false # Move the cross entropy unreduced loss calculation for lm head to fp16
|
||||
|
||||
# Megatron O2-style half-precision
|
||||
megatron_amp_O2: true # Enable O2-level automatic mixed precision using main parameters
|
||||
grad_allreduce_chunk_size_mb: 125
|
||||
|
||||
# Fusion
|
||||
grad_div_ar_fusion: true # Fuse grad division into torch.distributed.all_reduce. Only used with O2 and no pipeline parallelism..
|
||||
gradient_accumulation_fusion: true # Fuse weight gradient accumulation to GEMMs. Only used with pipeline parallelism and O2.
|
||||
bias_activation_fusion: true # Use a kernel that fuses the bias addition from weight matrices with the subsequent activation function.
|
||||
bias_dropout_add_fusion: true # Use a kernel that fuses the bias addition, dropout and residual connection addition.
|
||||
masked_softmax_fusion: true # Use a kernel that fuses the attention softmax with it's mask.
|
||||
apply_rope_fusion: true # Use a kernel to add rotary positional embeddings. Only used if position_embedding_type=rope
|
||||
cross_entropy_loss_fusion: true
|
||||
|
||||
# Miscellaneous
|
||||
seed: 1234
|
||||
resume_from_checkpoint: null # manually set the checkpoint file to load from
|
||||
use_cpu_initialization: false # Init weights on the CPU (slow for large models)
|
||||
onnx_safe: false # Use work-arounds for known problems with Torch ONNX exporter.
|
||||
apex_transformer_log_level: 30 # Python logging level displays logs with severity greater than or equal to this
|
||||
gradient_as_bucket_view: true # PyTorch DDP argument. Allocate gradients in a contiguous bucket to save memory (less fragmentation and buffer memory)
|
||||
sync_batch_comm: false # Enable stream synchronization after each p2p communication between pipeline stages
|
||||
|
||||
## Activation Checkpointing
|
||||
# NeMo Megatron supports 'selective' activation checkpointing where only the memory intensive part of attention is checkpointed.
|
||||
# These memory intensive activations are also less compute intensive which makes activation checkpointing more efficient for LLMs (20B+).
|
||||
# See Reducing Activation Recomputation in Large Transformer Models: https://arxiv.org/abs/2205.05198 for more details.
|
||||
# 'full' will checkpoint the entire transformer layer.
|
||||
activations_checkpoint_granularity: null # 'selective' or 'full'
|
||||
activations_checkpoint_method: null # 'uniform', 'block'
|
||||
# 'uniform' divides the total number of transformer layers and checkpoints the input activation
|
||||
# of each chunk at the specified granularity. When used with 'selective', 'uniform' checkpoints all attention blocks in the model.
|
||||
# 'block' checkpoints the specified number of layers per pipeline stage at the specified granularity
|
||||
activations_checkpoint_num_layers: null
|
||||
# when using 'uniform' this creates groups of transformer layers to checkpoint. Usually set to 1. Increase to save more memory.
|
||||
# when using 'block' this this will checkpoint the first activations_checkpoint_num_layers per pipeline stage.
|
||||
num_micro_batches_with_partial_activation_checkpoints: null
|
||||
# This feature is valid only when used with pipeline-model-parallelism.
|
||||
# When an integer value is provided, it sets the number of micro-batches where only a partial number of Transformer layers get checkpointed
|
||||
# and recomputed within a window of micro-batches. The rest of micro-batches in the window checkpoint all Transformer layers. The size of window is
|
||||
# set by the maximum outstanding micro-batch backpropagations, which varies at different pipeline stages. The number of partial layers to checkpoint
|
||||
# per micro-batch is set by 'activations_checkpoint_num_layers' with 'activations_checkpoint_method' of 'block'.
|
||||
# This feature enables using activation checkpoint at a fraction of micro-batches up to the point of full GPU memory usage.
|
||||
activations_checkpoint_layers_per_pipeline: null
|
||||
# This feature is valid only when used with pipeline-model-parallelism.
|
||||
# When an integer value (rounded down when float is given) is provided, it sets the number of Transformer layers to skip checkpointing at later
|
||||
# pipeline stages. For example, 'activations_checkpoint_layers_per_pipeline' of 3 makes pipeline stage 1 to checkpoint 3 layers less than
|
||||
# stage 0 and stage 2 to checkpoint 6 layers less stage 0, and so on. This is possible because later pipeline stage
|
||||
# uses less GPU memory with fewer outstanding micro-batch backpropagations. Used with 'num_micro_batches_with_partial_activation_checkpoints',
|
||||
# this feature removes most of activation checkpoints at the last pipeline stage, which is the critical execution path.
|
||||
|
||||
## Transformer Engine
|
||||
transformer_engine: true
|
||||
fp8: false # enables fp8 in TransformerLayer forward
|
||||
fp8_e4m3: false # sets fp8_format = recipe.Format.E4M3
|
||||
fp8_hybrid: false # sets fp8_format = recipe.Format.HYBRID
|
||||
fp8_margin: 0 # scaling margin
|
||||
fp8_interval: 1 # scaling update interval
|
||||
fp8_amax_history_len: 1024 # Number of steps for which amax history is recorded per tensor
|
||||
fp8_amax_compute_algo: 'max' # 'most_recent' or 'max'. Algorithm for computing amax from history
|
||||
ub_tp_comm_overlap: false # do not turn on because of b/397797926
|
||||
use_flash_attention: true
|
||||
gc_interval: 100
|
||||
|
||||
## Offloading Activations/Weights to CPU
|
||||
cpu_offloading: false
|
||||
cpu_offloading_num_layers: ${sum:${.num_layers},-1} # This value should be between [1,num_layers-1] as we don't want to offload the final layer's activations and expose any offloading duration for the final layer
|
||||
cpu_offloading_activations: true
|
||||
cpu_offloading_weights: true
|
||||
|
||||
data:
|
||||
# Path to data must be specified by the user.
|
||||
# Supports List, String and Dictionary
|
||||
# List : can override from the CLI: "model.data.data_prefix=[.5,/raid/data/pile/my-gpt3_00_text_document,.5,/raid/data/pile/my-gpt3_01_text_document]",
|
||||
# Or see example below:
|
||||
# data_prefix:
|
||||
# - .5
|
||||
# - /raid/data/pile/my-gpt3_00_text_document
|
||||
# - .5
|
||||
# - /raid/data/pile/my-gpt3_01_text_document
|
||||
# Dictionary: can override from CLI "model.data.data_prefix"={"train":[1.0, /path/to/data], "validation":/path/to/data, "test":/path/to/test}
|
||||
# Or see example below:
|
||||
# "model.data.data_prefix: {train:[1.0,/path/to/data], validation:[/path/to/data], test:[/path/to/test]}"
|
||||
data_prefix: [1.0, /data/hfbpe_gpt_training_data_text_document]
|
||||
index_mapping_dir: null # path to save index mapping .npy files, by default will save in the same location as data_prefix
|
||||
data_impl: mmap
|
||||
splits_string: 900,50,50
|
||||
seq_length: ${model.encoder_seq_length}
|
||||
skip_warmup: true
|
||||
num_workers: 2
|
||||
dataloader_type: single # cyclic
|
||||
reset_position_ids: false # Reset position ids after end-of-document token
|
||||
reset_attention_mask: false # Reset attention mask after end-of-document token
|
||||
eod_mask_loss: false # Mask loss for the end of document tokens
|
||||
validation_drop_last: true # Set to false if the last partial validation samples is to be consumed
|
||||
no_seqlen_plus_one_input_tokens: false # Set to True to disable fetching (sequence length + 1) input tokens, instead get (sequence length) input tokens and mask the last token
|
||||
pad_samples_to_global_batch_size: false # Set to True if you want to pad the last partial batch with -1's to equal global batch size
|
||||
shuffle_documents: true # Set to False to disable documents shuffling. Sample index will still be shuffled
|
||||
|
||||
# Nsys profiling options
|
||||
nsys_profile:
|
||||
enabled: false
|
||||
start_step: 0 # Global batch to start profiling
|
||||
end_step: 1 # Global batch to end profiling
|
||||
ranks: [0] # Global rank IDs to profile
|
||||
gen_shape: false # Generate model and kernel details including input shapes
|
||||
|
||||
memory_profile:
|
||||
enabled: false
|
||||
start_step: 0
|
||||
end_step: 1
|
||||
ranks: [0]
|
||||
output_path: /data # Must be a dir
|
||||
|
||||
optim:
|
||||
name: distributed_fused_adam # E.g., fused_adam or set _target_: torch.optim.AdamW field
|
||||
lr: 2e-5
|
||||
weight_decay: 0.01
|
||||
betas:
|
||||
- 0.9
|
||||
- 0.98
|
||||
bucket_cap_mb: 125
|
||||
overlap_grad_sync: true
|
||||
overlap_param_sync: true
|
||||
contiguous_grad_buffer: true
|
||||
contiguous_param_buffer: true
|
||||
sched:
|
||||
name: CosineAnnealing
|
||||
warmup_steps: 400
|
||||
constant_steps: 0
|
||||
min_lr: 2e-6
|
||||
+26
@@ -0,0 +1,26 @@
|
||||
# Copyright 2024 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
steps:
|
||||
- name: 'gcr.io/cloud-builders/docker'
|
||||
args:
|
||||
- 'build'
|
||||
- '--tag=${_ARTIFACT_REGISTRY}/${_IMAGE_NAME}'
|
||||
- '--file=docker/vertex-dist-recipes.Dockerfile'
|
||||
- '.'
|
||||
automapSubstitutions: true
|
||||
env:
|
||||
- 'DOCKER_BUILDKIT=1'
|
||||
images:
|
||||
- '${_ARTIFACT_REGISTRY}/${_IMAGE_NAME}'
|
||||
+41
@@ -0,0 +1,41 @@
|
||||
diff --git a/nemo/collections/nlp/parts/megatron_trainer_builder.py b/nemo/collections/nlp/parts/megatron_trainer_builder.py
|
||||
index b2c85cde4..a3a9670c3 100644
|
||||
--- a/nemo/collections/nlp/parts/megatron_trainer_builder.py
|
||||
+++ b/nemo/collections/nlp/parts/megatron_trainer_builder.py
|
||||
@@ -19,6 +19,7 @@ from lightning_fabric.utilities.exceptions import MisconfigurationException
|
||||
from omegaconf import DictConfig
|
||||
from pytorch_lightning import Trainer
|
||||
from pytorch_lightning.callbacks import ModelSummary
|
||||
+from pytorch_lightning.callbacks import Callback
|
||||
from pytorch_lightning.plugins.environments import TorchElasticEnvironment
|
||||
|
||||
from nemo.collections.common.metrics.perf_metrics import FLOPsMeasurementCallback
|
||||
@@ -38,6 +39,23 @@ from nemo.utils.callbacks.dist_ckpt_io import (
|
||||
AsyncFinalizerCallback,
|
||||
DistributedCheckpointIO,
|
||||
)
|
||||
+from vmg.util.device_stats import gpu_stats_str
|
||||
+
|
||||
+class GpuStatsMon(Callback):
|
||||
+ def on_train_start(self, trainer, pl_module) -> None:
|
||||
+ rank=pl_module.global_rank
|
||||
+ print(f'train_start: {rank=} {gpu_stats_str()}', flush=True)
|
||||
+
|
||||
+ def on_train_batch_start(self, trainer, pl_module, batch, batch_idx) -> None:
|
||||
+ rank=pl_module.global_rank
|
||||
+ print(f'batch_start: {rank=} {gpu_stats_str()}', flush=True)
|
||||
+
|
||||
+ def on_train_batch_end(self, trainer, pl_module, outputs, batch, batch_idx) -> None:
|
||||
+ rank=pl_module.global_rank
|
||||
+ print(f'batch_end: {rank=} {gpu_stats_str()}', flush=True)
|
||||
|
||||
|
||||
class MegatronTrainerBuilder:
|
||||
@@ -178,6 +196,7 @@ class MegatronTrainerBuilder:
|
||||
if self.cfg.get('exp_manager', {}).get('log_tflops_per_sec_per_gpu', True):
|
||||
callbacks.append(FLOPsMeasurementCallback(self.cfg))
|
||||
|
||||
+ callbacks.append(GpuStatsMon())
|
||||
return callbacks
|
||||
|
||||
def create_trainer(self, callbacks=None) -> Trainer:
|
||||
+41
@@ -0,0 +1,41 @@
|
||||
diff -ruN old-datasets/blended_megatron_dataset_builder.py datasets/blended_megatron_dataset_builder.py
|
||||
--- old-datasets/blended_megatron_dataset_builder.py 2025-05-02 04:08:45.369199665 +0000
|
||||
+++ datasets/blended_megatron_dataset_builder.py 2025-05-02 04:10:47.369119891 +0000
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
import logging
|
||||
import math
|
||||
+import os
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from typing import Any, Callable, Iterable, List, Optional, Type, Union
|
||||
|
||||
@@ -353,7 +354,7 @@
|
||||
num_dataset_builder_threads = self.config.num_dataset_builder_threads
|
||||
|
||||
if torch.distributed.is_initialized():
|
||||
- rank = torch.distributed.get_rank()
|
||||
+ rank = int(os.getenv("LOCAL_RANK", "0"))
|
||||
# First, build on rank 0
|
||||
if rank == 0:
|
||||
num_workers = num_dataset_builder_threads
|
||||
@@ -475,7 +476,7 @@
|
||||
Optional[Union[DistributedDataset, Iterable]]: The DistributedDataset instantion, the Iterable instantiation, or None
|
||||
"""
|
||||
if torch.distributed.is_initialized():
|
||||
- rank = torch.distributed.get_rank()
|
||||
+ rank = int(os.getenv("LOCAL_RANK", "0"))
|
||||
|
||||
dataset = None
|
||||
|
||||
diff -ruN old-datasets/gpt_dataset.py datasets/gpt_dataset.py
|
||||
--- old-datasets/gpt_dataset.py 2025-05-02 04:08:45.369199665 +0000
|
||||
+++ datasets/gpt_dataset.py 2025-05-02 04:09:30.309170278 +0000
|
||||
@@ -351,7 +351,7 @@
|
||||
|
||||
if not path_to_cache or (
|
||||
not cache_hit
|
||||
- and (not torch.distributed.is_initialized() or torch.distributed.get_rank() == 0)
|
||||
+ and (not torch.distributed.is_initialized() or int(os.getenv("LOCAL_RANK", "0")) == 0)
|
||||
):
|
||||
|
||||
log_single_rank(
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
diff --git a/scripts/checkpoint_converters/convert_llama_nemo_to_hf.py b/scripts/checkpoint_converters/convert_llama_nemo_to_hf.py
|
||||
index 8da15148d..005cae6c9 100644
|
||||
--- a/scripts/checkpoint_converters/convert_llama_nemo_to_hf.py
|
||||
+++ b/scripts/checkpoint_converters/convert_llama_nemo_to_hf.py
|
||||
@@ -104,6 +104,8 @@ def convert(input_nemo_file, output_hf_file, precision=None, cpu_only=False) ->
|
||||
dummy_trainer = Trainer(devices=1, accelerator='cpu', strategy=NLPDDPStrategy())
|
||||
model_config = MegatronGPTModel.restore_from(input_nemo_file, trainer=dummy_trainer, return_config=True)
|
||||
model_config.tensor_model_parallel_size = 1
|
||||
+ model_config.virtual_pipeline_model_parallel_size = None
|
||||
+ model_config.sequence_parallel = False
|
||||
model_config.pipeline_model_parallel_size = 1
|
||||
if cpu_only:
|
||||
map_location = torch.device('cpu')
|
||||
+24
@@ -0,0 +1,24 @@
|
||||
diff --git a/examples/nlp/language_modeling/tuning/megatron_gpt_finetuning.py b/examples/nlp/language_modeling/tuning/megatron_gpt_finetuning.py
|
||||
index bfe8ea359..dfeaf93b5 100644
|
||||
--- a/examples/nlp/language_modeling/tuning/megatron_gpt_finetuning.py
|
||||
+++ b/examples/nlp/language_modeling/tuning/megatron_gpt_finetuning.py
|
||||
@@ -13,6 +13,8 @@
|
||||
# limitations under the License.
|
||||
|
||||
import torch.multiprocessing as mp
|
||||
+import torch.distributed as dist
|
||||
+
|
||||
from omegaconf.omegaconf import OmegaConf
|
||||
|
||||
from nemo.collections.nlp.models.language_modeling.megatron_gpt_sft_model import MegatronGPTSFTModel
|
||||
@@ -76,6 +78,10 @@ def main(cfg) -> None:
|
||||
|
||||
trainer.fit(model)
|
||||
|
||||
+ if dist.is_available() and dist.is_initialized():
|
||||
+ dist.barrier()
|
||||
+ dist.destroy_process_group()
|
||||
+
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
diff --git a/src/utils/training_metrics/process_training_results.py b/src/utils/training_metrics/process_training_results.py
|
||||
index 3e82a66..e61e1d8 100644
|
||||
--- a/src/utils/training_metrics/process_training_results.py
|
||||
+++ b/src/utils/training_metrics/process_training_results.py
|
||||
@@ -134,7 +134,7 @@ def get_average_step_time(file: str, start_step: int, end_step: int) -> float:
|
||||
for line in datajson:
|
||||
if line.get("step") != "PARAMETER":
|
||||
step = line.get("step")
|
||||
- if step >= start_step and step <= end_step:
|
||||
+ if step >= start_step and step <= end_step and "train_step_timing in s" in line["data"]:
|
||||
time_step_accumulator += line["data"].get("train_step_timing in s")
|
||||
num_steps += 1
|
||||
if num_steps == 0:
|
||||
+10
@@ -0,0 +1,10 @@
|
||||
dllogger@git+https://github.com/NVIDIA/dllogger@v1.0.0
|
||||
|
||||
# Fixing these libraries versions to avoid conflicting or broken packages.
|
||||
immutabledict==4.2.1
|
||||
protobuf==4.25.8
|
||||
opencv-python-headless==4.11.0.86
|
||||
docutils==0.16
|
||||
urllib3==2.6.0
|
||||
google-cloud-storage==3.0.0
|
||||
retrying
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# cuml-cu12==24.8.0 was installed in nemo:24.09
|
||||
# Removing cuml=24.4.0 to avoid conflicting packages.
|
||||
cudf==24.4.0
|
||||
cugraph==24.4.0
|
||||
cugraph-service-server==24.4.0
|
||||
cuml==24.4.0
|
||||
dask-cudf==24.4.0
|
||||
raft-dask==24.4.0
|
||||
cugraph-dgl==24.4.0
|
||||
cugraph-pyg==24.4.0
|
||||
# The following packages are removed temporarily to avoid conflicting packages
|
||||
# and can be brought back if needed.
|
||||
tensorrt-llm==0.12.0
|
||||
img2dataset==1.45.0
|
||||
Sphinx==8.1.3
|
||||
sphinxcontrib-bibtex==2.6.3
|
||||
torchx==0.7.0
|
||||
nemo-run
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
# Dockerfile wrapping NeMo.
|
||||
#
|
||||
# To workaround base nemo docker image using too many layers, we use Multi-stage
|
||||
# build to first collect the additional files we'll need.
|
||||
FROM alpine:latest AS prep_files
|
||||
WORKDIR /workspace
|
||||
RUN mkdir -p configs vdt vdt/util
|
||||
COPY scripts/*.py vdt/
|
||||
COPY scripts/util/*.py vdt/util/
|
||||
COPY configs/* configs/
|
||||
COPY docker/patches/24.09/* vdt/patches/
|
||||
RUN chmod a+rwX -R vdt
|
||||
# Copy license.
|
||||
RUN wget https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/LICENSE
|
||||
|
||||
# Available tags
|
||||
# https://catalog.ngc.nvidia.com/orgs/nvidia/containers/nemo/tags
|
||||
# It installs NeMo source code in /opt/NeMo folder, with tag=r2.0.0
|
||||
FROM nvcr.io/nvidia/nemo:24.09
|
||||
|
||||
RUN apt-get update && apt-get install -y sudo zsh tmux && \
|
||||
rm -rf /var/lib/apt/lists*
|
||||
|
||||
RUN echo "deb [signed-by=/usr/share/keyrings/cloud.google.gpg] http://packages.cloud.google.com/apt cloud-sdk main" | \
|
||||
tee -a /etc/apt/sources.list.d/google-cloud-sdk.list && \
|
||||
curl https://packages.cloud.google.com/apt/doc/apt-key.gpg | \
|
||||
apt-key --keyring /usr/share/keyrings/cloud.google.gpg add - && \
|
||||
apt-get update -y && apt-get install google-cloud-sdk -y && \
|
||||
rm -rf /var/lib/apt/lists*
|
||||
|
||||
# Install libraries with pip
|
||||
ENV PIP_ROOT_USER_ACTION=ignore
|
||||
|
||||
# We expect this will be run in the root directory of the vertex-dist-recipes repo
|
||||
ARG HOST_SRC_DIR="."
|
||||
|
||||
# The pre-installed NeMo introduces a lot of deps conflicts.
|
||||
# We uninstall the confilicting libs and reinstall some of them as needed.
|
||||
COPY ${HOST_SRC_DIR}/docker/uninstall.txt /tmp/uninstall.txt
|
||||
RUN cat /tmp/uninstall.txt | grep -v '#' | xargs pip uninstall -y
|
||||
COPY ${HOST_SRC_DIR}/docker/requirements.txt /tmp/requirements.txt
|
||||
RUN pip install -r /tmp/requirements.txt
|
||||
|
||||
# Make sure there's no inconsistent pip libraries.
|
||||
RUN pip check
|
||||
|
||||
WORKDIR /workspace
|
||||
|
||||
# Copy configs
|
||||
COPY ${HOST_SRC_DIR}/configs/* /opt/NeMo/examples/nlp/language_modeling/conf/
|
||||
|
||||
# Copy all additional files we need from `prep_files` image.
|
||||
COPY --from=prep_files /workspace/ .
|
||||
|
||||
# Install for `src/utils/training_metrics/process_training_results.py` to report
|
||||
# throughput and MFU numbers.
|
||||
RUN git clone https://github.com/AI-Hypercomputer/gpu-recipes.git
|
||||
|
||||
# This hack is needed for multi-node training while not using a sharing file system.
|
||||
RUN patch --verbose -l -d /opt/megatron-lm/megatron/core/datasets -p1 -i /workspace/vdt/patches/local_rank.patch; \
|
||||
git -C /workspace/gpu-recipes apply /workspace/vdt/patches/throughput_calc.patch; \
|
||||
git -C /opt/NeMo apply /workspace/vdt/patches/nemo2hf.patch; \
|
||||
git -C /opt/NeMo apply /workspace/vdt/patches/sigabort.patch;
|
||||
# git -C /opt/NeMo apply /workspace/vdt/patches/gpu_stats.patch;
|
||||
|
||||
# Do not put an entrypoint here. Specify the entrypoint in the docker run script.
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
{
|
||||
"project_id": "<your_project_id>",
|
||||
"region": "us-central1",
|
||||
"zone": "us-central1-c",
|
||||
"bucket": "<your_bucket",
|
||||
"dataset_bucket": "github-repo/data/third-party/enwiki-latest-pages-articles",
|
||||
"image_uri": "<your_image_uri>",
|
||||
"strategy": "spot",
|
||||
"nodes": "2",
|
||||
"machine_type": "a3-megagpu-8g",
|
||||
"gpu_type": "NVIDIA_H100_MEGA_80GB",
|
||||
"gpus_per_node": "8",
|
||||
"recipe_name": "llama3_1_8b_pretrain_a3mega",
|
||||
"job_prefix": "vertex-ai",
|
||||
"reservation_name": ""
|
||||
}
|
||||
+49
@@ -0,0 +1,49 @@
|
||||
absl-py==2.2.2
|
||||
annotated-types==0.7.0
|
||||
anyio==4.9.0
|
||||
black==25.1.0
|
||||
cachetools==5.5.2
|
||||
certifi==2025.4.26
|
||||
charset-normalizer==3.4.2
|
||||
click==8.1.8
|
||||
docstring_parser==0.16
|
||||
google-api-core==2.24.2
|
||||
google-auth==2.40.1
|
||||
google-cloud-aiplatform==1.92.0
|
||||
google-cloud-bigquery==3.31.0
|
||||
google-cloud-core==2.4.3
|
||||
google-cloud-resource-manager==1.14.2
|
||||
google-cloud-storage==2.19.0
|
||||
google-crc32c==1.7.1
|
||||
google-genai==1.14.0
|
||||
google-resumable-media==2.7.2
|
||||
googleapis-common-protos==1.70.0
|
||||
grpc-google-iam-v1==0.14.2
|
||||
grpcio==1.71.0
|
||||
grpcio-status==1.71.0
|
||||
h11==0.16.0
|
||||
httpcore==1.0.9
|
||||
httpx==0.28.1
|
||||
idna==3.10
|
||||
mypy_extensions==1.1.0
|
||||
numpy==2.2.5
|
||||
packaging==25.0
|
||||
pathspec==0.12.1
|
||||
platformdirs==4.3.8
|
||||
proto-plus==1.26.1
|
||||
protobuf==5.29.4
|
||||
pyasn1==0.6.1
|
||||
pyasn1_modules==0.4.2
|
||||
pydantic==2.11.4
|
||||
pydantic_core==2.33.2
|
||||
python-dateutil==2.9.0.post0
|
||||
pytz==2025.2
|
||||
requests==2.32.4
|
||||
rsa==4.9.1
|
||||
shapely==2.1.0
|
||||
six==1.17.0
|
||||
sniffio==1.3.1
|
||||
typing-inspection==0.4.0
|
||||
typing_extensions==4.13.2
|
||||
urllib3==2.6.0
|
||||
websockets==15.0.1
|
||||
+173
@@ -0,0 +1,173 @@
|
||||
"""Launch script for Vertex distributed training"""
|
||||
|
||||
# Copy the sample_job_config.json file to job_config.json
|
||||
# to define the job parameters.
|
||||
#
|
||||
# Run like this:
|
||||
#
|
||||
# python3 vertex_dist_train/launch.py --config_file=job_config.json
|
||||
#
|
||||
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
import pprint
|
||||
from collections.abc import Sequence
|
||||
from typing import Any, List
|
||||
|
||||
from absl import app, flags
|
||||
from google.cloud import aiplatform
|
||||
from google.cloud.aiplatform_v1.types.custom_job import Scheduling
|
||||
from pytz import timezone
|
||||
|
||||
FLAGS = flags.FLAGS
|
||||
flags.DEFINE_string("config_file", None, "Path to JSON config file")
|
||||
flags.DEFINE_boolean(
|
||||
"debug", False, "Debug mode: just print the command, don't run it."
|
||||
)
|
||||
|
||||
|
||||
def launch_job(
|
||||
job_name: str,
|
||||
project: str,
|
||||
region: str,
|
||||
gcs_bucket: str,
|
||||
image_uri: str,
|
||||
entrypoint_cmd: List[str],
|
||||
trainer_args: List[Any],
|
||||
num_nodes: int,
|
||||
machine_type: str,
|
||||
num_gpus_per_node: int,
|
||||
gpu_type: str,
|
||||
strategy: str,
|
||||
reservation_name: str = "",
|
||||
):
|
||||
assert strategy in ("dws", "spot", "reservation")
|
||||
aiplatform.init(
|
||||
project=project, location=region, staging_bucket=gcs_bucket
|
||||
)
|
||||
|
||||
train_job = aiplatform.CustomContainerTrainingJob(
|
||||
display_name=job_name,
|
||||
container_uri=image_uri,
|
||||
command=entrypoint_cmd,
|
||||
)
|
||||
|
||||
job_args = dict(
|
||||
args=trainer_args,
|
||||
enable_web_access=True,
|
||||
replica_count=num_nodes,
|
||||
machine_type=machine_type,
|
||||
accelerator_type=gpu_type,
|
||||
accelerator_count=num_gpus_per_node,
|
||||
boot_disk_size_gb=1000,
|
||||
restart_job_on_worker_restart=True,
|
||||
#restart_job_on_worker_restart=False,
|
||||
)
|
||||
|
||||
if strategy == "spot":
|
||||
job_args.update({"scheduling_strategy": Scheduling.Strategy.SPOT.name})
|
||||
elif strategy == "dws":
|
||||
job_args.update(
|
||||
{"scheduling_strategy": Scheduling.Strategy.FLEX_START.name}
|
||||
)
|
||||
elif strategy == "reservation":
|
||||
assert reservation_name != "", (
|
||||
"If using a reservation, provide the reservation_name in the "
|
||||
"format `projects/{project_id_or_number}/zones/{zone}/"
|
||||
"reservations/{reservation_name}`"
|
||||
)
|
||||
job_args.update(
|
||||
{
|
||||
"reservation_affinity_type": "SPECIFIC_RESERVATION",
|
||||
"reservation_affinity_key": "compute.googleapis.com/reservation-name",
|
||||
"reservation_affinity_values": [reservation_name],
|
||||
}
|
||||
)
|
||||
|
||||
pprint.pprint(job_args)
|
||||
if not FLAGS.debug:
|
||||
train_job.submit(**job_args)
|
||||
|
||||
|
||||
def main(argv: Sequence[str]) -> None:
|
||||
config_file_path = FLAGS.config_file
|
||||
print(f"Reading job config from {config_file_path}")
|
||||
with open(config_file_path, encoding="utf-8") as config_file:
|
||||
config = json.load(config_file)
|
||||
|
||||
project_id = config["project_id"]
|
||||
region = config["region"]
|
||||
zone = config["zone"]
|
||||
bucket = config["bucket"]
|
||||
dataset_bucket = config["dataset_bucket"]
|
||||
n_nodes = int(config["nodes"])
|
||||
machine_type = config["machine_type"]
|
||||
num_gpus_per_node = int(config["gpus_per_node"])
|
||||
gpu_type = config["gpu_type"]
|
||||
reservation_name = config.get("reservation_name")
|
||||
reservation_full_name = (
|
||||
f"projects/{project_id}/zones/{zone}/reservations/{reservation_name}"
|
||||
if "reservation_name" in config
|
||||
else ""
|
||||
)
|
||||
|
||||
strategy = config["strategy"]
|
||||
recipe_name = config["recipe_name"]
|
||||
job_prefix = config["job_prefix"]
|
||||
image_uri = config["image_uri"]
|
||||
|
||||
# Job name
|
||||
timestamp = (
|
||||
datetime.datetime.now()
|
||||
.astimezone(timezone("US/Pacific"))
|
||||
.strftime("%Y%m%d_%H%M%S")
|
||||
)
|
||||
job_name = f"{recipe_name}-{timestamp}"
|
||||
if job_prefix:
|
||||
job_name = f"{job_prefix}-{job_name}"
|
||||
|
||||
base_output_dir = os.path.join("/gcs", bucket, job_name)
|
||||
|
||||
# Training command and args
|
||||
entrypoint_cmd = ["python3", "vdt/run.py"]
|
||||
|
||||
dataset_bucket = f"gs://{config['dataset_bucket']}"
|
||||
|
||||
trainer_args = [
|
||||
f"--train_data_gcs={dataset_bucket}",
|
||||
"/opt/NeMo/examples/nlp/language_modeling/megatron_gpt_pretraining.py",
|
||||
"--config-path=conf/",
|
||||
f"--config-name={recipe_name}.yaml",
|
||||
f"exp_manager.explicit_log_dir={base_output_dir}",
|
||||
f"exp_manager.dllogger_logger_kwargs.json_file={base_output_dir}/dllogger.json",
|
||||
"+exp_manager.create_tensorboard_logger=true",
|
||||
"exp_manager.create_checkpoint_callback=false",
|
||||
f"trainer.num_nodes={n_nodes}",
|
||||
f"trainer.devices={num_gpus_per_node}",
|
||||
"trainer.max_steps=10",
|
||||
"trainer.log_every_n_steps=1",
|
||||
"model.tokenizer.vocab_file=/data/gpt2-vocab.json",
|
||||
"model.tokenizer.merge_file=/data/gpt2-merges.txt",
|
||||
"model.data.data_prefix=[1.0,/data/hfbpe_gpt_training_data_text_document]",
|
||||
]
|
||||
|
||||
launch_job(
|
||||
job_name=job_name,
|
||||
project=project_id,
|
||||
region=region,
|
||||
gcs_bucket=bucket,
|
||||
image_uri=image_uri,
|
||||
entrypoint_cmd=entrypoint_cmd,
|
||||
trainer_args=trainer_args,
|
||||
num_nodes=n_nodes,
|
||||
machine_type=machine_type,
|
||||
num_gpus_per_node=num_gpus_per_node,
|
||||
gpu_type=gpu_type,
|
||||
strategy=strategy,
|
||||
reservation_name=reservation_full_name,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app.run(main)
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
"""Entrypoint for Vertex Distributed Training container."""
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Sequence
|
||||
from subprocess import STDOUT, check_output, run
|
||||
|
||||
from absl import app, flags, logging
|
||||
from util import cluster_spec
|
||||
|
||||
from retrying import retry
|
||||
|
||||
# PyTorch barrier call which synchronizes all of the nodes before launching the training process.
|
||||
# This makes sure that processes will block until all processes are ready.
|
||||
# Improves the reliability of spot VM usage for multi-node training jobs
|
||||
|
||||
@retry(stop_max_attempt_number=100, wait_exponential_multiplier=1000)
|
||||
def barrier_with_retry() -> None:
|
||||
import torch
|
||||
logging.info("Starting barrier on RANK {}".format(os.environ["RANK"]))
|
||||
torch.distributed.init_process_group()
|
||||
torch.distributed.barrier()
|
||||
torch.distributed.destroy_process_group()
|
||||
logging.info("Finished barrier on RANK {}".format(os.environ["RANK"]))
|
||||
|
||||
def main(unused_argv: Sequence[str]) -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument(
|
||||
"--train_data_gcs",
|
||||
type=str,
|
||||
help="Download training data from gcs path",
|
||||
)
|
||||
args, unknown = parser.parse_known_args()
|
||||
|
||||
for key, val in os.environ.items():
|
||||
logging.info("ENV %s=%s", key, val)
|
||||
|
||||
if args.train_data_gcs:
|
||||
local_dir = "/data"
|
||||
if not os.path.exists(local_dir):
|
||||
os.mkdir(local_dir)
|
||||
logging.info("downloading %s to %s...", args.train_data_gcs, local_dir)
|
||||
check_output(
|
||||
[
|
||||
"gcloud",
|
||||
"storage",
|
||||
"cp",
|
||||
"-r",
|
||||
f"{args.train_data_gcs}/*",
|
||||
local_dir,
|
||||
],
|
||||
stderr=STDOUT,
|
||||
)
|
||||
logging.info("%s downloaded.", args.train_data_gcs)
|
||||
|
||||
primary_node_addr, primary_node_port, node_rank, num_nodes = (
|
||||
cluster_spec.get_cluster_spec()
|
||||
)
|
||||
|
||||
cmd = [
|
||||
"torchrun",
|
||||
"--nproc-per-node=8",
|
||||
f"--nnodes={num_nodes}",
|
||||
f"--node_rank={node_rank}",
|
||||
]
|
||||
if num_nodes > 1:
|
||||
cmd += [
|
||||
"--max-restarts=3",
|
||||
"--rdzv-backend=static",
|
||||
f'--rdzv_id={os.getenv("CLOUD_ML_JOB_ID", primary_node_port)}',
|
||||
f"--rdzv-endpoint={primary_node_addr}:{primary_node_port}",
|
||||
]
|
||||
cmd += unknown
|
||||
|
||||
logging.info("launching with cmd: \n%s", " \\\n".join(cmd))
|
||||
barrier_with_retry()
|
||||
run(cmd, stdout=sys.stdout, stderr=sys.stdout, check=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
logging.get_absl_handler().python_handler.stream = sys.stdout
|
||||
app.run(
|
||||
main, flags_parser=lambda _args: flags.FLAGS(_args, known_only=True)
|
||||
)
|
||||
+81
@@ -0,0 +1,81 @@
|
||||
"""Get cluster info from environment variables."""
|
||||
|
||||
import dataclasses
|
||||
import json
|
||||
import os
|
||||
|
||||
from absl import logging
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class ClusterInfo:
|
||||
"""Contains information about the cluster.
|
||||
|
||||
Attributes:
|
||||
primary_node_addr: The address of the primary node.
|
||||
primary_node_port: The port of the primary node.
|
||||
node_rank: The rank of the node.
|
||||
num_nodes: The number of nodes in the cluster.
|
||||
"""
|
||||
|
||||
primary_node_addr: str | None = None
|
||||
primary_node_port: str | None = None
|
||||
node_rank: int = 0
|
||||
num_nodes: int = 1
|
||||
|
||||
# Allows unpacking operation like
|
||||
# primary_node_addr, primary_node_port, _, _ = ClusterInfo()
|
||||
# See https://stackoverflow.com/a/70753113
|
||||
def __iter__(self):
|
||||
return iter(dataclasses.astuple(self))
|
||||
|
||||
|
||||
def get_cluster_spec() -> ClusterInfo:
|
||||
"""Parses CLUSTER_SPEC environment variable and returns the cluster info.
|
||||
|
||||
Returns:
|
||||
A ClusterInfo object.
|
||||
"""
|
||||
cluster_spec = os.getenv("CLUSTER_SPEC", None)
|
||||
|
||||
# If CLUSTER_SPEC is not set, use individual vars to construct cluster info.
|
||||
if not cluster_spec:
|
||||
cluster_info = ClusterInfo(
|
||||
primary_node_addr=os.getenv("MASTER_ADDR", None),
|
||||
primary_node_port=os.getenv("MASTER_PORT", None),
|
||||
node_rank=int(os.getenv("RANK", "0")),
|
||||
num_nodes=int(os.getenv("NNODES", "1")),
|
||||
)
|
||||
return cluster_info
|
||||
|
||||
cluster_data = json.loads(cluster_spec)
|
||||
# Get primary node info
|
||||
primary_node = cluster_data["cluster"]["workerpool0"][0]
|
||||
logging.info("primary node: %s", primary_node)
|
||||
primary_node_addr, primary_node_port = primary_node.split(":")
|
||||
logging.info("primary node address: %s", primary_node_addr)
|
||||
logging.info("primary node port: %s", primary_node_port)
|
||||
|
||||
# Determine node rank of this machine
|
||||
workerpool = cluster_data["task"]["type"]
|
||||
if workerpool == "workerpool0":
|
||||
node_rank = 0
|
||||
elif workerpool == "workerpool1":
|
||||
# Add 1 for the primary node, since `index` is the index of workerpool1.
|
||||
node_rank = cluster_data["task"]["index"] + 1
|
||||
else:
|
||||
raise ValueError(
|
||||
"Only workerpool0 and workerpool1 are supported. Unknown workerpool:"
|
||||
f" {workerpool}"
|
||||
)
|
||||
logging.info("node rank: %s", node_rank)
|
||||
|
||||
# Calculate total nodes.
|
||||
num_nodes = 1 # For the primary node.
|
||||
if "workerpool1" in cluster_data["cluster"]:
|
||||
num_nodes += len(cluster_data["cluster"]["workerpool1"])
|
||||
logging.info("num nodes: %s", num_nodes)
|
||||
|
||||
return ClusterInfo(
|
||||
primary_node_addr, primary_node_port, node_rank, num_nodes
|
||||
)
|
||||
+59
@@ -0,0 +1,59 @@
|
||||
"""Add tests for cluster_spec.py."""
|
||||
|
||||
import os
|
||||
|
||||
from . import cluster_spec
|
||||
|
||||
|
||||
# TODO(styer): Use pytest instead
|
||||
class ClusterSpecTest(googletest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.curr_env_var = os.environ.copy()
|
||||
|
||||
def tearDown(self):
|
||||
super().tearDown()
|
||||
os.environ = self.curr_env_var
|
||||
|
||||
def test_get_cluster_spec_from_env_vars(self):
|
||||
os.environ["CLUSTER_SPEC"] = ""
|
||||
os.environ["MASTER_ADDR"] = "127.0.0.1"
|
||||
os.environ["MASTER_PORT"] = "8080"
|
||||
os.environ["RANK"] = "0"
|
||||
os.environ["NNODES"] = "2"
|
||||
cluster_info = cluster_spec.get_cluster_spec()
|
||||
self.assertEqual(cluster_info.primary_node_addr, "127.0.0.1")
|
||||
self.assertEqual(cluster_info.primary_node_port, "8080")
|
||||
self.assertEqual(cluster_info.node_rank, 0)
|
||||
self.assertEqual(cluster_info.num_nodes, 2)
|
||||
|
||||
def test_get_cluster_spec_from_cluster_spec(self):
|
||||
os.environ[
|
||||
"CLUSTER_SPEC"
|
||||
] = """
|
||||
{
|
||||
"cluster": {
|
||||
"workerpool0": [
|
||||
"127.0.0.1:8080"
|
||||
],
|
||||
"workerpool1": [
|
||||
"127.0.0.2:8080",
|
||||
"127.0.0.3:8080"
|
||||
]
|
||||
},
|
||||
"task": {
|
||||
"type": "workerpool1",
|
||||
"index": 0
|
||||
}
|
||||
}
|
||||
"""
|
||||
cluster_info = cluster_spec.get_cluster_spec()
|
||||
self.assertEqual(cluster_info.primary_node_addr, "127.0.0.1")
|
||||
self.assertEqual(cluster_info.primary_node_port, "8080")
|
||||
self.assertEqual(cluster_info.node_rank, 1)
|
||||
self.assertEqual(cluster_info.num_nodes, 3)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
googletest.main()
|
||||
+7
-7
@@ -66,7 +66,7 @@ mkdir -p "$local_folder"
|
||||
mkdir -p "$output_folder"
|
||||
|
||||
# Download the content from the GCS URI
|
||||
gsutil -m cp -r "$gcs_dataset_path"/* "$local_folder/"
|
||||
gcloud storage cp --recursive "$gcs_dataset_path"/* "$local_folder/"
|
||||
|
||||
# Process files in the local folder
|
||||
for file in "$local_folder"/*; do
|
||||
@@ -122,23 +122,23 @@ cp -r "$output_folder" "$images_folder"/images_2
|
||||
pushd "$images_folder"/images_2
|
||||
ls | xargs -P 8 -I {} mogrify -resize 50% {}
|
||||
popd
|
||||
gsutil -m cp -r "$images_folder"/images_2/* "$gcs_experiment_path"/data/images_2
|
||||
gcloud storage cp --recursive "$images_folder"/images_2/* "$gcs_experiment_path"/data/images_2
|
||||
|
||||
cp -r "$output_folder" "$images_folder"/images_4
|
||||
pushd "$images_folder"/images_4
|
||||
ls | xargs -P 8 -I {} mogrify -resize 25% {}
|
||||
popd
|
||||
gsutil -m cp -r "$images_folder"/images_4/* "$gcs_experiment_path"/data/images_4
|
||||
gcloud storage cp --recursive "$images_folder"/images_4/* "$gcs_experiment_path"/data/images_4
|
||||
|
||||
cp -r "$output_folder" "$images_folder"/images_8
|
||||
pushd "$images_folder"/images_8
|
||||
ls | xargs -P 8 -I {} mogrify -resize 12.5% {}
|
||||
popd
|
||||
gsutil -m cp "$images_folder"/images_8/* "$gcs_experiment_path"/data/images_8
|
||||
gcloud storage cp "$images_folder"/images_8/* "$gcs_experiment_path"/data/images_8
|
||||
|
||||
# Copy images and sparse reconstruction files to gcs experiment folder.
|
||||
gsutil -m cp "$images_folder"/images/* "$gcs_experiment_path"/data/images
|
||||
gsutil -m cp -r "$local_folder"/sparse "$gcs_experiment_path"/data
|
||||
gsutil -m cp "$local_folder"/database.db "$gcs_experiment_path"/data
|
||||
gcloud storage cp "$images_folder"/images/* "$gcs_experiment_path"/data/images
|
||||
gcloud storage cp --recursive "$local_folder"/sparse "$gcs_experiment_path"/data
|
||||
gcloud storage cp "$local_folder"/database.db "$gcs_experiment_path"/data
|
||||
|
||||
echo "Processing complete."
|
||||
@@ -99,14 +99,14 @@ create_dir_if_not_exists "$CHECKPOINTS_PATH"
|
||||
touch "$local_experiment_path/$exp_folder_name/log_render.txt"
|
||||
|
||||
# Copy experiment from GCS bucket to local
|
||||
gsutil -m cp -r "${args[-gcs_experiment_path]}/data" "$local_experiment_path/$exp_folder_name" || exit 1
|
||||
gsutil -m cp -r "${args[-gcs_experiment_path]}/checkpoints/${training_job_name}/*" "$CHECKPOINTS_PATH" || exit 1
|
||||
gcloud storage cp --recursive "${args[-gcs_experiment_path]}/data" "$local_experiment_path/$exp_folder_name" || exit 1
|
||||
gcloud storage cp --recursive "${args[-gcs_experiment_path]}/checkpoints/${training_job_name}/*" "$CHECKPOINTS_PATH" || exit 1
|
||||
|
||||
# Check and copy keyframes file.
|
||||
if [[ -n ${args[-gcs_keyframes_file]} ]]; then
|
||||
keyframes_file_basename=$(basename "${args[-gcs_keyframes_file]}")
|
||||
local_keyframes_file="$local_dataset_path/$keyframes_file_basename"
|
||||
gsutil cp "${args[-gcs_keyframes_file]}" "$local_keyframes_file" || exit 1
|
||||
gcloud storage cp "${args[-gcs_keyframes_file]}" "$local_keyframes_file" || exit 1
|
||||
echo "Local keyframe file: $local_keyframes_file"
|
||||
launch_rendering "$local_keyframes_file"
|
||||
else
|
||||
@@ -114,4 +114,4 @@ else
|
||||
fi
|
||||
|
||||
# Copy rendered data back to GCS.
|
||||
gsutil -m cp -r "$OUTPUT_RENDER_PATH" "${args[-gcs_experiment_path]}/render/${rendering_job_name}"
|
||||
gcloud storage cp --recursive "$OUTPUT_RENDER_PATH" "${args[-gcs_experiment_path]}/render/${rendering_job_name}"
|
||||
@@ -74,7 +74,7 @@ create_dir_if_not_exists "$local_experiment_path"
|
||||
create_dir_if_not_exists "$local_experiment_path/$scene_folder_name"
|
||||
|
||||
# Copy experiment from GCS bucket to local.
|
||||
gsutil -m cp -r "${gcs_experiment_path}/data" "$local_experiment_path/$scene_folder_name" || exit 1
|
||||
gcloud storage cp --recursive "${gcs_experiment_path}/data" "$local_experiment_path/$scene_folder_name" || exit 1
|
||||
|
||||
echo "GCS Experiment: $gcs_experiment_path"
|
||||
echo "Gin Config File: $gin_config_file"
|
||||
@@ -89,6 +89,6 @@ accelerate launch train.py --gin_configs="$gin_config_file" \
|
||||
--gin_bindings="Config.factor = ${factor}" \
|
||||
--gin_bindings="Config.max_steps = ${max_training_steps}"
|
||||
|
||||
gsutil -m rm -r "${gcs_experiment_path}/checkpoints/${training_job_name}"
|
||||
gsutil -m cp -r "$local_experiment_path/$scene_folder_name/config.gin" "${gcs_experiment_path}/${training_job_name}_config.gin"
|
||||
gsutil -m cp -r "$local_experiment_path/$scene_folder_name/checkpoints/*/*" "${gcs_experiment_path}/checkpoints/${training_job_name}"
|
||||
gcloud storage rm --recursive "${gcs_experiment_path}/checkpoints/${training_job_name}"
|
||||
gcloud storage cp --recursive "$local_experiment_path/$scene_folder_name/config.gin" "${gcs_experiment_path}/${training_job_name}_config.gin"
|
||||
gcloud storage cp --recursive "$local_experiment_path/$scene_folder_name/checkpoints/*/*" "${gcs_experiment_path}/checkpoints/${training_job_name}"
|
||||
@@ -1,13 +1,16 @@
|
||||
"""Common util functions for notebook."""
|
||||
|
||||
import base64
|
||||
from collections.abc import Sequence
|
||||
import datetime
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
from typing import Any, Dict, Sequence
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
from google import auth
|
||||
from google.cloud import storage
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
@@ -281,7 +284,7 @@ def decode_image(
|
||||
return image
|
||||
|
||||
|
||||
def get_label_map(label_map_yaml_filepath: str) -> Dict[int, str]:
|
||||
def get_label_map(label_map_yaml_filepath: str) -> dict[int, str]:
|
||||
"""Returns class id to label mapping given a filepath to the label map.
|
||||
|
||||
Args:
|
||||
@@ -331,6 +334,7 @@ def vqa_predict(
|
||||
image: Any,
|
||||
language_code: str = "en",
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> Sequence[str]:
|
||||
"""Predicts the answer to a question about an image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
@@ -354,7 +358,9 @@ def vqa_predict(
|
||||
"image": resized_image_base64,
|
||||
})
|
||||
|
||||
response = endpoint.predict(instances=instances)
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return [pred.get("response") for pred in response.predictions]
|
||||
|
||||
|
||||
@@ -364,6 +370,7 @@ def caption_predict(
|
||||
image: Any,
|
||||
caption_prompt: bool = False,
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> str:
|
||||
"""Predicts a caption for a given image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
@@ -378,7 +385,9 @@ def caption_predict(
|
||||
instance["prompt"] = caption_prompt_format.format(language_code)
|
||||
|
||||
instances = [instance]
|
||||
response = endpoint.predict(instances=instances)
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return response.predictions[0].get("response")
|
||||
|
||||
|
||||
@@ -387,6 +396,7 @@ def ocr_predict(
|
||||
ocr_prompt: str,
|
||||
image: Any,
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> str:
|
||||
"""Extracts text from a given image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
@@ -398,7 +408,9 @@ def ocr_predict(
|
||||
instance["prompt"] = ocr_prompt
|
||||
instances = [instance]
|
||||
|
||||
response = endpoint.predict(instances=instances)
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return response.predictions[0].get("response")
|
||||
|
||||
|
||||
@@ -407,6 +419,7 @@ def detect_predict(
|
||||
detect_prompt: str,
|
||||
image: Any,
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> str:
|
||||
"""Predicts the answer to a question about an image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
@@ -418,7 +431,9 @@ def detect_predict(
|
||||
instance["prompt"] = detect_prompt
|
||||
instances = [instance]
|
||||
|
||||
response = endpoint.predict(instances=instances)
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return response.predictions[0].get("response")
|
||||
|
||||
|
||||
@@ -495,6 +510,17 @@ def get_quota(project_id: str, region: str, resource_id: str) -> int:
|
||||
):
|
||||
return -1
|
||||
all_regions_data = quota_data[0]["consumerQuotaLimits"][0]["quotaBuckets"]
|
||||
|
||||
# If the quota data does not have dimensions, it is global quota. However,
|
||||
# global quota may be overridden by regional quota. So we need to check the
|
||||
# global quota first.
|
||||
global_quota = -1
|
||||
if (
|
||||
all_regions_data
|
||||
and "dimensions" not in all_regions_data[0]
|
||||
and "effectiveLimit" in all_regions_data[0]
|
||||
):
|
||||
global_quota = int(all_regions_data[0]["effectiveLimit"])
|
||||
for region_data in all_regions_data:
|
||||
if (
|
||||
region_data.get("dimensions")
|
||||
@@ -504,12 +530,13 @@ def get_quota(project_id: str, region: str, resource_id: str) -> int:
|
||||
return int(region_data["effectiveLimit"])
|
||||
else:
|
||||
return 0
|
||||
return -1
|
||||
return global_quota
|
||||
|
||||
|
||||
def get_resource_id(
|
||||
accelerator_type: str,
|
||||
is_for_training: bool,
|
||||
is_spot: bool = False,
|
||||
is_restricted_image: bool = False,
|
||||
is_dynamic_workload_scheduler: bool = False,
|
||||
) -> str:
|
||||
@@ -519,6 +546,7 @@ def get_resource_id(
|
||||
accelerator_type: The accelerator type.
|
||||
is_for_training: Whether the resource is used for training. Set false for
|
||||
serving use case.
|
||||
is_spot: Whether the resource is used with Spot.
|
||||
is_restricted_image: Whether the image is hosted in `vertex-ai-restricted`.
|
||||
is_dynamic_workload_scheduler: Whether the resource is used with Dynamic
|
||||
Workload Scheduler.
|
||||
@@ -534,7 +562,9 @@ def get_resource_id(
|
||||
"NVIDIA_A100_80GB": "nvidia_a100_80gb_gpus",
|
||||
"NVIDIA_H100_80GB": "nvidia_h100_gpus",
|
||||
"NVIDIA_H100_MEGA_80GB": "nvidia_h100_mega_gpus",
|
||||
"NVIDIA_H200_141GB": "nvidia_h200_gpus",
|
||||
"NVIDIA_TESLA_T4": "nvidia_t4_gpus",
|
||||
"TPU_V6e": "tpu_v6e",
|
||||
"TPU_V5e": "tpu_v5e",
|
||||
"TPU_V3": "tpu_v3",
|
||||
}
|
||||
@@ -549,6 +579,10 @@ def get_resource_id(
|
||||
restricted_image_training_accelerator_map = {
|
||||
"NVIDIA_A100_80GB": "restricted_image_training_nvidia_a100_80gb_gpus",
|
||||
}
|
||||
spot_serving_accelerator_map = {
|
||||
key: f"custom_model_serving_preemptible_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
}
|
||||
serving_accelerator_map = {
|
||||
key: f"custom_model_serving_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
@@ -577,8 +611,11 @@ def get_resource_id(
|
||||
else:
|
||||
if is_dynamic_workload_scheduler:
|
||||
raise ValueError("Dynamic Workload Scheduler does not work for serving.")
|
||||
if accelerator_type in serving_accelerator_map:
|
||||
return serving_accelerator_map[accelerator_type]
|
||||
accelerator_map = (
|
||||
spot_serving_accelerator_map if is_spot else serving_accelerator_map
|
||||
)
|
||||
if accelerator_type in accelerator_map:
|
||||
return accelerator_map[accelerator_type]
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Could not find accelerator type: {accelerator_type} for serving."
|
||||
@@ -591,13 +628,28 @@ def check_quota(
|
||||
accelerator_type: str,
|
||||
accelerator_count: int,
|
||||
is_for_training: bool,
|
||||
is_spot: bool = False,
|
||||
is_restricted_image: bool = False,
|
||||
is_dynamic_workload_scheduler: bool = False,
|
||||
):
|
||||
"""Checks if the project and the region has the required quota."""
|
||||
) -> None:
|
||||
"""Checks if the project and the region has the required quota.
|
||||
|
||||
Args:
|
||||
project_id: The project id.
|
||||
region: The region.
|
||||
accelerator_type: The accelerator type.
|
||||
accelerator_count: The number of accelerators to check quota for.
|
||||
is_for_training: Whether the resource is used for training. Set false for
|
||||
serving use case.
|
||||
is_spot: Whether the resource is used with Spot.
|
||||
is_restricted_image: Whether the image is hosted in `vertex-ai-restricted`.
|
||||
is_dynamic_workload_scheduler: Whether the resource is used with Dynamic
|
||||
Workload Scheduler.
|
||||
"""
|
||||
resource_id = get_resource_id(
|
||||
accelerator_type,
|
||||
is_for_training=is_for_training,
|
||||
is_spot=is_spot,
|
||||
is_restricted_image=is_restricted_image,
|
||||
is_dynamic_workload_scheduler=is_dynamic_workload_scheduler,
|
||||
)
|
||||
@@ -634,3 +686,62 @@ def get_deploy_source() -> str:
|
||||
# Legacy workbench, legacy colab, or other custom environments.
|
||||
return "notebook_environment_unspecified"
|
||||
|
||||
|
||||
def _is_operation_done(op_name: str, region: str) -> bool:
|
||||
"""Checks if the operation is done.
|
||||
|
||||
Args:
|
||||
op_name: The name of the operation to poll.
|
||||
region: The region of the operation.
|
||||
|
||||
Returns:
|
||||
True if the operation is done, False otherwise.
|
||||
|
||||
Raises:
|
||||
ValueError: If the operation failed.
|
||||
"""
|
||||
creds, _ = auth.default()
|
||||
auth_req = auth.transport.requests.Request()
|
||||
creds.refresh(auth_req)
|
||||
headers = {
|
||||
"Authorization": f"Bearer {creds.token}",
|
||||
}
|
||||
url = f"https://{region}-aiplatform.googleapis.com/ui/{op_name}"
|
||||
response = requests.get(url, headers=headers)
|
||||
operation_data = response.json()
|
||||
if "error" in operation_data:
|
||||
raise ValueError(f"Operation failed: {operation_data['error']}")
|
||||
return operation_data.get("done", False)
|
||||
|
||||
|
||||
def poll_and_wait(
|
||||
op_name: str, region: str, total_wait: int, interval: int = 60
|
||||
) -> None:
|
||||
"""Polls the operation and waits for it to complete.
|
||||
|
||||
Args:
|
||||
op_name: The name of the operation to poll.
|
||||
region: The region of the operation.
|
||||
total_wait: The total wait time in seconds.
|
||||
interval: The interval between each poll in seconds.
|
||||
|
||||
Raises:
|
||||
TimeoutError: If the operation times out.
|
||||
"""
|
||||
start_time = time.time()
|
||||
while True:
|
||||
if _is_operation_done(op_name, region):
|
||||
break
|
||||
time_elapsed = time.time() - start_time
|
||||
if time_elapsed > total_wait:
|
||||
raise TimeoutError(
|
||||
f"Operation timed out after {int(time_elapsed)} seconds."
|
||||
)
|
||||
print(
|
||||
"\rStill waiting for operation... Elapsed time in seconds:"
|
||||
f" {int(time_elapsed):<6}",
|
||||
end="",
|
||||
flush=True,
|
||||
)
|
||||
time.sleep(interval)
|
||||
|
||||
|
||||
+50
-23
@@ -7,7 +7,7 @@ import json
|
||||
import multiprocessing
|
||||
import os
|
||||
import subprocess
|
||||
from typing import Any, Callable, Dict, Union
|
||||
from typing import Any, Callable, Dict, Tuple, Union
|
||||
from absl import logging
|
||||
import accelerate
|
||||
import datasets
|
||||
@@ -70,7 +70,9 @@ def force_gcs_fuse_path(gcs_uri: str) -> str:
|
||||
|
||||
|
||||
def download_gcs_uri_to_local(
|
||||
gcs_uri: str, destination_dir: str = LOCAL_BASE_MODEL_DIR
|
||||
gcs_uri: str,
|
||||
destination_dir: str = LOCAL_BASE_MODEL_DIR,
|
||||
check_path_exists: bool = True,
|
||||
) -> str:
|
||||
"""Downloads GCS URI to local.
|
||||
|
||||
@@ -81,6 +83,7 @@ def download_gcs_uri_to_local(
|
||||
Args:
|
||||
gcs_uri: GCS URI to download.
|
||||
destination_dir: Local directory directory.
|
||||
check_path_exists: Whether to check if the path exists.
|
||||
|
||||
Returns:
|
||||
Local path to target folder/file.
|
||||
@@ -89,7 +92,7 @@ def download_gcs_uri_to_local(
|
||||
destination_dir,
|
||||
os.path.basename(os.path.normpath(gcs_uri)),
|
||||
)
|
||||
if os.path.exists(target):
|
||||
if check_path_exists and os.path.exists(target):
|
||||
logging.info("File %s already exists.", target)
|
||||
return target
|
||||
if accelerate.PartialState().is_local_main_process:
|
||||
@@ -99,10 +102,10 @@ def download_gcs_uri_to_local(
|
||||
if not os.path.exists(destination_dir):
|
||||
os.mkdir(destination_dir)
|
||||
subprocess.check_output([
|
||||
"gsutil",
|
||||
"-m",
|
||||
"gcloud",
|
||||
"storage",
|
||||
"cp",
|
||||
"-r",
|
||||
"--recursive",
|
||||
gcs_uri,
|
||||
destination_dir,
|
||||
])
|
||||
@@ -415,13 +418,42 @@ def get_filtered_dataset(
|
||||
return filtered_dataset
|
||||
|
||||
|
||||
def format_dataset(
|
||||
dataset: datasets.Dataset,
|
||||
input_column: str,
|
||||
template: str = None,
|
||||
tokenizer: transformers.PreTrainedTokenizer | None = None,
|
||||
) -> datasets.Dataset:
|
||||
"""Takes a raw dataset and formats it using a template and tokenizer.
|
||||
|
||||
Args:
|
||||
dataset: The raw (unprocessed) dataset to format.
|
||||
input_column: The input column in the dataset to be used or updaded by the
|
||||
template. If it does not exist, the template's `prompt_no_input` will be
|
||||
used, and the input_column will be created.
|
||||
template: Name of the JSON template file under `templates/` or GCS path to
|
||||
the template file.
|
||||
tokenizer: The tokenizer to use for chat_template templates.
|
||||
|
||||
Returns:
|
||||
A dataset compatible with the template.
|
||||
"""
|
||||
return dataset.map(
|
||||
_format_template_fn(
|
||||
template,
|
||||
input_column=input_column,
|
||||
tokenizer=tokenizer,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def load_dataset_with_template(
|
||||
dataset_name: str,
|
||||
split: str,
|
||||
input_column: str,
|
||||
template: str = None,
|
||||
tokenizer: transformers.PreTrainedTokenizer | None = None,
|
||||
) -> Any:
|
||||
) -> Tuple[Any, Any]:
|
||||
"""Loads dataset with templates.
|
||||
|
||||
Args:
|
||||
@@ -435,19 +467,15 @@ def load_dataset_with_template(
|
||||
tokenizer: The tokenizer to use for chat_template templates.
|
||||
|
||||
Returns:
|
||||
A dataset compatible with the template.
|
||||
The raw dataset and the dataset compatible with the template.
|
||||
"""
|
||||
dataset = _get_dataset(dataset_name, split=split)
|
||||
raw = _get_dataset(dataset_name, split=split)
|
||||
if template:
|
||||
dataset = dataset.map(
|
||||
_format_template_fn(
|
||||
template,
|
||||
input_column=input_column,
|
||||
tokenizer=tokenizer,
|
||||
)
|
||||
)
|
||||
templated = format_dataset(raw, input_column, template, tokenizer)
|
||||
else:
|
||||
templated = None
|
||||
|
||||
return dataset
|
||||
return raw, templated
|
||||
|
||||
|
||||
def validate_dataset_with_template(
|
||||
@@ -521,12 +549,11 @@ def validate_dataset_with_template(
|
||||
f" https://github.com/GoogleCloudPlatform/{_VERTEX_AI_SAMPLES_GITHUB_REPO_NAME}/tree/main/{_VERTEX_AI_SAMPLES_GITHUB_TEMPLATE_DIR}."
|
||||
)
|
||||
|
||||
dataset = _get_dataset(dataset_name, split, num_proc).map(
|
||||
_format_template_fn(
|
||||
template_path,
|
||||
input_column=input_column,
|
||||
tokenizer=tokenizer,
|
||||
)
|
||||
dataset = format_dataset(
|
||||
_get_dataset(dataset_name, split, num_proc),
|
||||
input_column,
|
||||
template_path,
|
||||
tokenizer,
|
||||
)
|
||||
|
||||
if tokenizer is not None:
|
||||
|
||||
+8
@@ -32,6 +32,7 @@ class DockerCommandBuilder(CommandBuilder):
|
||||
super().__init__()
|
||||
self._docker_uri = [docker_uri]
|
||||
self.privilege_mode = []
|
||||
self.entrypoint = []
|
||||
|
||||
self._defaults = [
|
||||
'docker',
|
||||
@@ -62,6 +63,9 @@ class DockerCommandBuilder(CommandBuilder):
|
||||
def add_privilege_mode(self):
|
||||
self.privilege_mode = ['--privileged']
|
||||
|
||||
def add_entrypoint(self, entrypoint: list[str]):
|
||||
self.entrypoint = entrypoint
|
||||
|
||||
def build_cmd(self) -> str:
|
||||
return (
|
||||
self._defaults
|
||||
@@ -69,6 +73,7 @@ class DockerCommandBuilder(CommandBuilder):
|
||||
+ self._mount_maps
|
||||
+ self.privilege_mode
|
||||
+ self._docker_uri
|
||||
+ self.entrypoint
|
||||
)
|
||||
|
||||
|
||||
@@ -85,3 +90,6 @@ class PythonCommandBuilder(CommandBuilder):
|
||||
def build_cmd(self) -> str:
|
||||
os.environ.update(self._env_vars)
|
||||
return self._defaults
|
||||
|
||||
def add_entrypoint(self, entrypoint: list[str]):
|
||||
self._defaults = entrypoint
|
||||
+51
-20
@@ -3,6 +3,7 @@
|
||||
import copy
|
||||
import dataclasses
|
||||
import datetime
|
||||
import inspect
|
||||
import os
|
||||
import signal
|
||||
import subprocess
|
||||
@@ -11,7 +12,7 @@ from absl import flags
|
||||
from absl import logging
|
||||
from absl.testing import parameterized
|
||||
import command_builder
|
||||
import frozendict
|
||||
import immutabledict
|
||||
import torch
|
||||
|
||||
_DOCKER_URI = flags.DEFINE_string('docker_uri', None, 'docker image uri')
|
||||
@@ -33,19 +34,19 @@ _LOCAL_OUTPUT_DIR = flags.DEFINE_string(
|
||||
|
||||
_GCS_INPUT_DIR = flags.DEFINE_string(
|
||||
'gcs_input_dir',
|
||||
'gs://peft-docker-test',
|
||||
'gs://vmg-tuning-docker-test',
|
||||
'GCS directory that stores model checkpoint, dataset and etc.',
|
||||
)
|
||||
|
||||
_GCS_OUTPUT_DIR = flags.DEFINE_string(
|
||||
'gcs_output_dir',
|
||||
'gs://peft-docker-test/output',
|
||||
'gs://vmg-tuning-docker-test/output',
|
||||
'GCS directory that stores test output.',
|
||||
)
|
||||
|
||||
_GCS_TESTDATA_DIR = 'peft-train-image-test'
|
||||
|
||||
_THROUGHPUT_TEST_EXCEPTIONS = frozendict.frozendict({
|
||||
_THROUGHPUT_TEST_EXCEPTIONS = immutabledict.immutabledict({
|
||||
('bm_deepspeed_zero3_8gpu_gemma-2-9b-it_4bit.txt', '12.0'): float('inf'),
|
||||
('bm_fsdp_8gpu_llama3.1-70b-hf_4bit.txt', '20.0'): float('inf'),
|
||||
('bm_deepspeed_zero2_8gpu_gemma-2-2b-it_bfloat16.txt', '12.0'): 20.0,
|
||||
@@ -101,26 +102,25 @@ class TestBase(parameterized.TestCase):
|
||||
return self.command_builder.build_cmd() + self.task_cmd_builder.build_cmd()
|
||||
|
||||
def run_cmd(self) -> int:
|
||||
logging.info('running command: \n%s', ' \\\n'.join(self.cmd()))
|
||||
if _DRY_RUN.value:
|
||||
return 0
|
||||
|
||||
p = subprocess.Popen(self.cmd(), stdout=sys.stdout, stderr=sys.stderr)
|
||||
try:
|
||||
unused_output, unused_error = p.communicate()
|
||||
return p.returncode
|
||||
except KeyboardInterrupt:
|
||||
p.send_signal(signal.SIGINT)
|
||||
return 0
|
||||
return run_cmd(self.cmd(), output_file=None)
|
||||
|
||||
def gcs_output_dir(self):
|
||||
return _GCS_OUTPUT_DIR.value
|
||||
|
||||
def local_input_dir(self):
|
||||
"""Returns local input dir in host/docker."""
|
||||
return _LOCAL_INPUT_DIR.value
|
||||
|
||||
def local_output_dir(self):
|
||||
"""Returns local output dir in host/docker."""
|
||||
return _LOCAL_OUTPUT_DIR.value
|
||||
|
||||
def local_input_dir(self):
|
||||
return _LOCAL_INPUT_DIR.value
|
||||
def get_testcase_name(self):
|
||||
"""Returns the function name at the calling site."""
|
||||
# https://docs.python.org/3/library/inspect.html#inspect.FrameInfo
|
||||
cur_frame = inspect.currentframe()
|
||||
# https://stackoverflow.com/a/17366561
|
||||
return cur_frame.f_back.f_code.co_name
|
||||
|
||||
|
||||
def get_timestamp():
|
||||
@@ -157,13 +157,44 @@ def get_test_data_path(name: str, download: bool = True) -> str:
|
||||
|
||||
local_data = os.path.join(_LOCAL_INPUT_DIR.value, name)
|
||||
if not os.path.exists(local_data):
|
||||
download_from_gcs(
|
||||
os.path.join(_GCS_INPUT_DIR.value, name), _LOCAL_INPUT_DIR.value
|
||||
)
|
||||
# If `name` is a file in sub-folders, then create the sub-folders under
|
||||
# `_LOCAL_INPUT_DIR`.
|
||||
local_data_dir = os.path.dirname(local_data)
|
||||
if not os.path.exists(local_data_dir):
|
||||
os.makedirs(local_data_dir)
|
||||
|
||||
download_from_gcs(os.path.join(_GCS_INPUT_DIR.value, name), local_data_dir)
|
||||
|
||||
return local_data
|
||||
|
||||
|
||||
def run_cmd(cmd: list[str], output_file: str = None) -> int:
|
||||
"""Runs the command and returns the return code.
|
||||
|
||||
Args:
|
||||
cmd: The command to run.
|
||||
output_file: The file to write the output to.
|
||||
|
||||
Returns:
|
||||
The return code of the command.
|
||||
"""
|
||||
logging.info('running command: \n%s', ' \\\n'.join(cmd))
|
||||
if _DRY_RUN.value:
|
||||
return 0
|
||||
stdout = sys.stdout if output_file is None else open(output_file, 'w')
|
||||
p = subprocess.Popen(cmd, stdout=stdout, stderr=sys.stderr)
|
||||
try:
|
||||
unused_output, unused_error = p.communicate()
|
||||
return_code = p.returncode
|
||||
except KeyboardInterrupt:
|
||||
p.send_signal(signal.SIGINT)
|
||||
return_code = 0
|
||||
finally:
|
||||
if output_file is not None:
|
||||
stdout.close()
|
||||
return return_code
|
||||
|
||||
|
||||
def get_pretrained_model_name_or_path(model_id: str) -> str:
|
||||
# If `model_id` contains `/`, it is assumed to be HF model or model from GCS.
|
||||
if '/' in model_id:
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
"""Get cluster info from environment variables."""
|
||||
|
||||
import dataclasses
|
||||
import json
|
||||
import os
|
||||
|
||||
from absl import logging
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class ClusterInfo:
|
||||
"""Contains information about the cluster.
|
||||
|
||||
Attributes:
|
||||
primary_node_addr: The address of the primary node.
|
||||
primary_node_port: The port of the primary node.
|
||||
node_rank: The rank of the node.
|
||||
num_nodes: The number of nodes in the cluster.
|
||||
"""
|
||||
|
||||
primary_node_addr: str | None = None
|
||||
primary_node_port: str | None = None
|
||||
node_rank: int = 0
|
||||
num_nodes: int = 1
|
||||
|
||||
# Allows unpacking operation like
|
||||
# primary_node_addr, primary_node_port, _, _ = ClusterInfo()
|
||||
# See https://stackoverflow.com/a/70753113
|
||||
def __iter__(self):
|
||||
return iter(dataclasses.astuple(self))
|
||||
|
||||
|
||||
def get_cluster_spec() -> ClusterInfo:
|
||||
"""Parses CLUSTER_SPEC environment variable and returns the cluster info.
|
||||
|
||||
Returns:
|
||||
A ClusterInfo object.
|
||||
"""
|
||||
cluster_spec = os.getenv('CLUSTER_SPEC', None)
|
||||
|
||||
# If CLUSTER_SPEC is not set, use individual vars to construct cluster info.
|
||||
if not cluster_spec:
|
||||
cluster_info = ClusterInfo(
|
||||
primary_node_addr=os.getenv('MASTER_ADDR', None),
|
||||
primary_node_port=os.getenv('MASTER_PORT', None),
|
||||
node_rank=int(os.getenv('RANK', '0')),
|
||||
num_nodes=int(os.getenv('NNODES', '1')),
|
||||
)
|
||||
return cluster_info
|
||||
|
||||
cluster_data = json.loads(cluster_spec)
|
||||
# Get primary node info
|
||||
primary_node = cluster_data['cluster']['workerpool0'][0]
|
||||
logging.info('primary node: %s', primary_node)
|
||||
primary_node_addr, primary_node_port = primary_node.split(':')
|
||||
logging.info('primary node address: %s', primary_node_addr)
|
||||
logging.info('primary node port: %s', primary_node_port)
|
||||
|
||||
# Determine node rank of this machine
|
||||
workerpool = cluster_data['task']['type']
|
||||
if workerpool == 'workerpool0':
|
||||
node_rank = 0
|
||||
elif workerpool == 'workerpool1':
|
||||
# Add 1 for the primary node, since `index` is the index of workerpool1.
|
||||
node_rank = cluster_data['task']['index'] + 1
|
||||
else:
|
||||
raise ValueError(
|
||||
'Only workerpool0 and workerpool1 are supported. Unknown workerpool:'
|
||||
f' {workerpool}'
|
||||
)
|
||||
logging.info('node rank: %s', node_rank)
|
||||
|
||||
# Calculate total nodes.
|
||||
num_nodes = 1 # For the primary node.
|
||||
if 'workerpool1' in cluster_data['cluster']:
|
||||
num_nodes += len(cluster_data['cluster']['workerpool1'])
|
||||
logging.info('num nodes: %s', num_nodes)
|
||||
|
||||
return ClusterInfo(primary_node_addr, primary_node_port, node_rank, num_nodes)
|
||||
@@ -0,0 +1,24 @@
|
||||
"""Utility functions."""
|
||||
|
||||
import logging
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
def run_cmd(cmd: list[str]) -> float:
|
||||
"""Runs the command and logs the output.
|
||||
|
||||
Args:
|
||||
cmd: The command to run.
|
||||
|
||||
Returns:
|
||||
The time it took to run the command.
|
||||
"""
|
||||
cmd_str = ' \\\n'.join(cmd)
|
||||
logging.info('launching cmd: \n%s', cmd_str)
|
||||
start_time = time.time()
|
||||
subprocess.run(cmd, stdout=sys.stdout, stderr=sys.stdout, check=True)
|
||||
elapsed_time = round(time.time() - start_time, 2)
|
||||
logging.info('Command %s finished in %0.2f seconds.', cmd_str, elapsed_time)
|
||||
return elapsed_time
|
||||
@@ -0,0 +1,197 @@
|
||||
"""Calculate dataset statistics like token, example and character counts."""
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
import dataclasses
|
||||
import json
|
||||
from typing import Any
|
||||
import datasets
|
||||
import numpy as np
|
||||
import transformers
|
||||
from util import dataset_validation_util
|
||||
|
||||
_MAX_NUM_DATASET_SAMPLES = 6
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class SupervisedTuningDatasetBucket:
|
||||
"""Represents a histogram bucket for tuning dataset distribution stats."""
|
||||
|
||||
count: float = 0
|
||||
left: float = 0
|
||||
right: float = 0
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class SupervisedTuningDatasetDistribution:
|
||||
"""Represents a histogram with summary statistics for tuning dataset distribution stats."""
|
||||
|
||||
sum: int = 0
|
||||
billable_sum: int = 0
|
||||
min: float = 0
|
||||
max: float = 0
|
||||
mean: float = 0
|
||||
median: float = 0
|
||||
p5: float = 0
|
||||
p95: float = 0
|
||||
buckets: list[SupervisedTuningDatasetBucket] = dataclasses.field(
|
||||
default_factory=list
|
||||
)
|
||||
|
||||
|
||||
# Represents detailed tuning dataset statistics.
|
||||
@dataclasses.dataclass
|
||||
class SupervisedTuningDataStats:
|
||||
"""Represents detailed tuning dataset stats."""
|
||||
|
||||
tuning_dataset_example_count: int = 0
|
||||
total_tuning_character_count: int = 0
|
||||
total_billable_token_count: int = 0
|
||||
tuning_step_count: int = 0
|
||||
# Represents a histogram and some summary statistics of the number of input
|
||||
# tokens across examples.
|
||||
user_input_token_distribution: SupervisedTuningDatasetDistribution | None = (
|
||||
None
|
||||
)
|
||||
# Represents a histogram and some summary statistics for the number of output
|
||||
# tokens across examples.
|
||||
user_output_token_distribution: SupervisedTuningDatasetDistribution | None = (
|
||||
None
|
||||
)
|
||||
# Represents the number of "messages" (a single-turn conversation will have a
|
||||
# single message) across examples.
|
||||
user_message_per_example_distribution: (
|
||||
SupervisedTuningDatasetDistribution | None
|
||||
) = None
|
||||
user_dataset_examples: list[str] = dataclasses.field(default_factory=list)
|
||||
|
||||
|
||||
def get_dataset_stats(
|
||||
*,
|
||||
raw: Any,
|
||||
templated: Any,
|
||||
template: str,
|
||||
tokenizer: transformers.PreTrainedTokenizer,
|
||||
column: str,
|
||||
effective_batch_size: int,
|
||||
) -> Mapping[str, Any]:
|
||||
"""Calculates dataset statistics for managed fine-tuning, e.g., total number of tokens."""
|
||||
tokenized_dataset = templated.map(lambda x: tokenizer(x[column]))
|
||||
inputs = tokenized_dataset["input_ids"]
|
||||
tuning_dataset_example_count = int(len(inputs))
|
||||
total_billable_token_count = int(np.sum([len(ex) for ex in inputs]))
|
||||
total_tuning_character_count = int(
|
||||
np.sum([len(ex[column]) for ex in templated])
|
||||
)
|
||||
tuning_step_count = (
|
||||
tuning_dataset_example_count + effective_batch_size - 1
|
||||
) // effective_batch_size
|
||||
|
||||
# Assume that data is represented as ChatCompletions or Vertex Text-Bison
|
||||
# formats to extract per-example input/output tokens.
|
||||
user_inputs = []
|
||||
user_outputs = []
|
||||
user_input_messages_counts = []
|
||||
|
||||
for ex in raw:
|
||||
if "messages" in ex:
|
||||
messages = ex["messages"]
|
||||
if messages:
|
||||
# For ChatCompletions assume the last turn (i.e. the instruction
|
||||
# response) is the expected output.
|
||||
user_inputs.append({**ex, "messages": messages[:-1]})
|
||||
user_outputs.append({**ex, "messages": messages[-1:]})
|
||||
# Exclude everything but the last message for the number of input
|
||||
# messages.
|
||||
user_input_messages_counts.append(len(messages[:-1]))
|
||||
elif "input_text" in ex:
|
||||
# For Vertex Text-Bison, the `output_text` field is the expected output.
|
||||
user_inputs.append({**ex, "output_text": ""})
|
||||
user_outputs.append(
|
||||
{**ex, "input_text": ex["output_text"], "output_text": ""}
|
||||
)
|
||||
# Vertex Text-Bison goes from input -> output; i.e. there is only a single
|
||||
# input "message".
|
||||
user_input_messages_counts.append(1)
|
||||
|
||||
def calc_histogram(
|
||||
counts: Sequence[int],
|
||||
) -> SupervisedTuningDatasetDistribution:
|
||||
mean = np.mean(counts)
|
||||
median = np.median(counts).item()
|
||||
max_count = np.max(counts).item()
|
||||
min_count = np.min(counts).item()
|
||||
count_sum = np.sum(counts).item()
|
||||
p5 = np.percentile(counts, 0.05).item()
|
||||
p95 = np.percentile(counts, 0.95).item()
|
||||
hist, bin_edges = np.histogram(counts, bins=10)
|
||||
|
||||
return SupervisedTuningDatasetDistribution(
|
||||
sum=count_sum,
|
||||
billable_sum=count_sum,
|
||||
min=min_count,
|
||||
max=max_count,
|
||||
mean=mean,
|
||||
median=median,
|
||||
p5=p5,
|
||||
p95=p95,
|
||||
buckets=[
|
||||
SupervisedTuningDatasetBucket(
|
||||
count=hist[i].item(),
|
||||
left=bin_edges[i].item(),
|
||||
right=bin_edges[i + 1].item(),
|
||||
)
|
||||
for i in range(len(hist))
|
||||
],
|
||||
)
|
||||
|
||||
# Tokenize input and output messages separately to generate separate summary
|
||||
# statistics about them.
|
||||
user_input_token_distribution = None
|
||||
if user_inputs:
|
||||
user_input_dataset = dataset_validation_util.format_dataset(
|
||||
datasets.Dataset.from_list(user_inputs), column, template, tokenizer
|
||||
)
|
||||
user_input_tokenized_dataset = user_input_dataset.map(
|
||||
lambda x: tokenizer(x[column])
|
||||
)
|
||||
user_input_tokens = user_input_tokenized_dataset["input_ids"]
|
||||
user_input_token_counts = np.array([len(ex) for ex in user_input_tokens])
|
||||
user_input_token_distribution = calc_histogram(user_input_token_counts)
|
||||
|
||||
user_output_token_distribution = None
|
||||
if user_outputs:
|
||||
user_output_dataset = dataset_validation_util.format_dataset(
|
||||
datasets.Dataset.from_list(user_outputs), column, template, tokenizer
|
||||
)
|
||||
user_output_tokenized_dataset = user_output_dataset.map(
|
||||
lambda x: tokenizer(x[column])
|
||||
)
|
||||
user_output_tokens = user_output_tokenized_dataset["input_ids"]
|
||||
user_output_token_counts = np.array([len(ex) for ex in user_output_tokens])
|
||||
user_output_token_distribution = calc_histogram(user_output_token_counts)
|
||||
|
||||
user_messages_per_example_distribution = None
|
||||
if user_input_messages_counts:
|
||||
user_input_messages_counts = np.array(user_input_messages_counts)
|
||||
user_messages_per_example_distribution = calc_histogram(
|
||||
user_input_messages_counts
|
||||
)
|
||||
|
||||
user_dataset_examples = [
|
||||
json.dumps(ex)
|
||||
for ex in raw.shuffle().select(
|
||||
range(min(len(raw), _MAX_NUM_DATASET_SAMPLES))
|
||||
)
|
||||
]
|
||||
|
||||
dataset_stats = SupervisedTuningDataStats(
|
||||
tuning_dataset_example_count=tuning_dataset_example_count,
|
||||
total_tuning_character_count=total_tuning_character_count,
|
||||
total_billable_token_count=total_billable_token_count,
|
||||
tuning_step_count=tuning_step_count,
|
||||
user_input_token_distribution=user_input_token_distribution,
|
||||
user_output_token_distribution=user_output_token_distribution,
|
||||
user_message_per_example_distribution=user_messages_per_example_distribution,
|
||||
user_dataset_examples=user_dataset_examples,
|
||||
)
|
||||
return dataclasses.asdict(dataset_stats)
|
||||
@@ -0,0 +1,140 @@
|
||||
"""Util functions for reporting device (GPU, CPU) stats."""
|
||||
|
||||
import dataclasses
|
||||
|
||||
import psutil
|
||||
import pynvml
|
||||
import torch
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class GpuStats:
|
||||
"""Holds information about GPU usage stats.
|
||||
|
||||
For memory related, see
|
||||
https://pytorch.org/docs/stable/notes/cuda.html#cuda-memory-management
|
||||
"""
|
||||
|
||||
# device id
|
||||
device_id: int
|
||||
# memory reserved.
|
||||
reserved: float
|
||||
# memory occupied.
|
||||
occupied: float
|
||||
# memory reserved, but not used.
|
||||
unused: float
|
||||
# nvidia-smi usually reports more memory usages than pytorch (for driver,
|
||||
# kernel and etc). `smi_diff` tracks this difference.
|
||||
smi_diff: float
|
||||
# Gpu utilization.
|
||||
util: float
|
||||
|
||||
# Allows unpacking operation like
|
||||
# device_id, reserved, occupied, unused, smi_diff, util = GpuStats(...)
|
||||
# See https://stackoverflow.com/a/70753113
|
||||
def __iter__(self):
|
||||
return iter(dataclasses.astuple(self))
|
||||
|
||||
|
||||
def gpu_stats() -> GpuStats:
|
||||
"""Reports GPU memory usage and utilization."""
|
||||
# See https://pytorch.org/docs/stable/notes/cuda.html#memory-management
|
||||
bytes_per_gb = 1024.0**3
|
||||
device = torch.cuda.current_device()
|
||||
occupied = torch.cuda.memory_allocated(device) / bytes_per_gb
|
||||
reserved = torch.cuda.memory_reserved(device) / bytes_per_gb
|
||||
unused = reserved - occupied
|
||||
|
||||
def smi_mem(device):
|
||||
try:
|
||||
pynvml.nvmlInit()
|
||||
handle = pynvml.nvmlDeviceGetHandleByIndex(device)
|
||||
info = pynvml.nvmlDeviceGetMemoryInfo(handle)
|
||||
return info.used / bytes_per_gb
|
||||
except pynvml.NVMLError:
|
||||
return 0.0
|
||||
|
||||
mem_used_smi = smi_mem(device)
|
||||
smi_diff = mem_used_smi - reserved
|
||||
|
||||
util = torch.cuda.utilization(device)
|
||||
return GpuStats(device, reserved, occupied, unused, smi_diff, util)
|
||||
|
||||
|
||||
def gpu_stats_str(stats: GpuStats | None = None) -> str:
|
||||
if stats is None:
|
||||
stats = gpu_stats()
|
||||
device, reserved, occupied, unused, smi_diff, util = stats
|
||||
return (
|
||||
f"GPU ({device=}) memory: {reserved:.2f}({occupied=:.2f}, {unused=:.2f}),"
|
||||
f" {smi_diff=:.2f} GB. Utilization: {util:.2f}%"
|
||||
)
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class CpuStats:
|
||||
"""Holds information about CPU usage stats."""
|
||||
|
||||
# Total CPU virtual memory i.e. virtual memory allocated + unallocated.
|
||||
total_virtual_mem: float
|
||||
# CPU virtual memory available for use.
|
||||
unallocated_virtual_mem: float
|
||||
# CPU virtual memory already used.
|
||||
allocated_virtual_mem: float
|
||||
# Total CPU swap memory i.e. swap memory allocated + unallocated.
|
||||
total_swap_mem: float
|
||||
# CPU swap memory available for use.
|
||||
unallocated_swap_mem: float
|
||||
# CPU swap memory already used.
|
||||
allocated_swap_mem: float
|
||||
# CPU utilization percentage.
|
||||
utilization: float
|
||||
|
||||
|
||||
def cpu_stats() -> CpuStats:
|
||||
"""Reports CPU memory usage and utilization."""
|
||||
|
||||
# https://psutil.readthedocs.io/en/latest/#memory
|
||||
gb = 1024.0**3
|
||||
vmem = psutil.virtual_memory()
|
||||
vmem_total = vmem.total / gb
|
||||
vmem_available = vmem.available / gb
|
||||
vmem_used = vmem_total - vmem_available
|
||||
smem = psutil.swap_memory()
|
||||
swap_total = smem.total / gb
|
||||
swap_free = smem.free / gb
|
||||
swap_used = smem.used / gb
|
||||
# https://psutil.readthedocs.io/en/latest/#psutil.cpu_percent
|
||||
cpu_util = psutil.cpu_percent(interval=1e-6)
|
||||
return CpuStats(
|
||||
total_virtual_mem=vmem_total,
|
||||
unallocated_virtual_mem=vmem_available,
|
||||
allocated_virtual_mem=vmem_used,
|
||||
total_swap_mem=swap_total,
|
||||
unallocated_swap_mem=swap_free,
|
||||
allocated_swap_mem=swap_used,
|
||||
utilization=cpu_util,
|
||||
)
|
||||
|
||||
|
||||
def cpu_stats_str(stats: CpuStats | None = None) -> str:
|
||||
"""Returns a string representation of the CPU stats."""
|
||||
|
||||
if stats is None:
|
||||
stats = cpu_stats()
|
||||
total, occupied, unused = (
|
||||
stats.total_virtual_mem,
|
||||
stats.allocated_virtual_mem,
|
||||
stats.unallocated_virtual_mem,
|
||||
)
|
||||
virtual_mem = (
|
||||
f"CPU virtual memory: {total:.2f}({occupied=:.2f}, {unused=:.2f}) GB"
|
||||
)
|
||||
total, occupied, unused = (
|
||||
stats.total_swap_mem,
|
||||
stats.allocated_swap_mem,
|
||||
stats.unallocated_swap_mem,
|
||||
)
|
||||
swap_mem = f"CPU swap memory: {total:.2f}({occupied=:.2f}, {unused=:.2f}) GB"
|
||||
percent = stats.utilization
|
||||
return f"{virtual_mem} {swap_mem} CPU Utilization: {percent:.2f}%"
|
||||
@@ -11,7 +11,7 @@ from transformers.trainer_callback import TrainerCallback
|
||||
from transformers.trainer_callback import TrainerControl
|
||||
from transformers.trainer_callback import TrainerState
|
||||
|
||||
from vertex_vision_model_garden_peft.train.vmg import utils
|
||||
from util import device_stats
|
||||
|
||||
|
||||
class TrainerStatsCallback(TrainerCallback):
|
||||
@@ -75,13 +75,15 @@ class TrainerStatsCallback(TrainerCallback):
|
||||
state.global_step - 1
|
||||
)
|
||||
|
||||
gpu_stats = utils.gpu_stats()
|
||||
self._peak_mem = max(gpu_stats.total_mem, self._peak_mem)
|
||||
gpu_stats = device_stats.gpu_stats()
|
||||
self._peak_mem = max(
|
||||
gpu_stats.reserved + gpu_stats.smi_diff, self._peak_mem
|
||||
)
|
||||
logging.info(
|
||||
'on_step_end: Throughput: %.2f token/s. %s, %s',
|
||||
throughput,
|
||||
utils.gpu_stats_str(gpu_stats),
|
||||
utils.cpu_stats_str(),
|
||||
device_stats.gpu_stats_str(gpu_stats),
|
||||
device_stats.cpu_stats_str(),
|
||||
)
|
||||
|
||||
def on_train_begin(
|
||||
@@ -95,8 +97,8 @@ class TrainerStatsCallback(TrainerCallback):
|
||||
self._start_time = time.time()
|
||||
logging.info(
|
||||
'on_train_begin: %s, %s',
|
||||
utils.gpu_stats_str(),
|
||||
utils.cpu_stats_str(),
|
||||
device_stats.gpu_stats_str(),
|
||||
device_stats.cpu_stats_str(),
|
||||
)
|
||||
|
||||
def on_train_end(
|
||||
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
compute_environment: LOCAL_MACHINE
|
||||
debug: false
|
||||
distributed_type: FSDP
|
||||
downcast_bf16: 'no'
|
||||
enable_cpu_affinity: false
|
||||
fsdp_config:
|
||||
fsdp_auto_wrap_policy: TRANSFORMER_BASED_WRAP
|
||||
fsdp_transformer_layer_cls_to_wrap: Qwen2DecoderLayer
|
||||
fsdp_backward_prefetch: NO_PREFETCH
|
||||
fsdp_cpu_ram_efficient_loading: true
|
||||
fsdp_forward_prefetch: false
|
||||
fsdp_offload_params: true
|
||||
fsdp_sharding_strategy: FULL_SHARD
|
||||
fsdp_state_dict_type: SHARDED_STATE_DICT
|
||||
fsdp_sync_module_states: true
|
||||
fsdp_use_orig_params: false
|
||||
fsdp_activation_checkpointing: false
|
||||
main_training_function: main
|
||||
mixed_precision: bf16
|
||||
machine_rank: 0
|
||||
num_machines: 1
|
||||
num_processes: 8
|
||||
rdzv_backend: static
|
||||
same_network: true
|
||||
tpu_env: []
|
||||
tpu_use_cluster: false
|
||||
tpu_use_sudo: false
|
||||
use_cpu: false
|
||||
+1
@@ -16,6 +16,7 @@ diffusers==0.25.1
|
||||
evaluate==0.4.3
|
||||
fsspec==2024.3.1
|
||||
gcsfs==2024.3.1
|
||||
immutabledict==4.2.1
|
||||
ninja==1.11.1 # Needed to avoid `ninja 1.11.1.1 is not supported on this platform` error
|
||||
nltk==3.9.1
|
||||
optimum==1.17.1
|
||||
|
||||
+3
-1
@@ -68,10 +68,12 @@ RUN mkdir -p ./vertex_vision_model_garden_peft/
|
||||
COPY model_oss/peft/train/vmg/configs/* ./vertex_vision_model_garden_peft/
|
||||
COPY model_oss/peft/train/vmg/*.py ./vertex_vision_model_garden_peft/train/vmg/
|
||||
COPY model_oss/peft/train/vmg/templates /diffusers/examples/util/templates
|
||||
COPY model_oss/util /diffusers/examples/util
|
||||
COPY model_oss/peft/train/util/*.py /diffusers/examples/util/
|
||||
COPY model_oss/util/* /diffusers/examples/util/
|
||||
COPY model_oss/notebook_util/dataset_validation_util.py /diffusers/examples/util
|
||||
COPY model_oss/peft/train/vmg/tests/*.py ./vertex_vision_model_garden_peft/tests/
|
||||
COPY model_oss/peft/train/test_utils/test_util.py ./vertex_vision_model_garden_peft/tests/
|
||||
COPY model_oss/peft/train/test_utils/command_builder.py ./vertex_vision_model_garden_peft/tests/
|
||||
|
||||
RUN chmod a+rwX -R /diffusers/examples/
|
||||
ENV PYTHONPATH /diffusers/examples/
|
||||
|
||||
@@ -37,7 +37,6 @@ class EvalConfig:
|
||||
steps: The number of steps to run evaluation.
|
||||
tasks: The list of tasks to run evaluation on.
|
||||
per_device_batch_size: The per device batch size for evaluation.
|
||||
num_fewshot: The number of few-shot examples to use for evaluation.
|
||||
limit: The maximum number of examples to evaluate.
|
||||
metric_name: The name of the metric to compute.
|
||||
tokenize_dataset: Whether to tokenize the dataset.
|
||||
@@ -50,7 +49,6 @@ class EvalConfig:
|
||||
|
||||
steps: int
|
||||
per_device_batch_size: int
|
||||
num_fewshot: int | None
|
||||
limit: float | None
|
||||
metric_name: Sequence[str]
|
||||
tokenize_dataset: bool
|
||||
@@ -99,7 +97,7 @@ def create_trainer(
|
||||
kwargs["tokenizer"] = tokenizer
|
||||
|
||||
try:
|
||||
eval_dataset = dataset_validation_util.load_dataset_with_template(
|
||||
_, eval_dataset = dataset_validation_util.load_dataset_with_template(
|
||||
dataset_name=eval_config.dataset_path,
|
||||
split=eval_config.split,
|
||||
input_column=eval_config.column,
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
"""Instruct/Chat with LoRA models."""
|
||||
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
import dataclasses
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
@@ -23,6 +22,8 @@ import trl
|
||||
import wandb
|
||||
|
||||
from util import dataset_validation_util
|
||||
from util import dataset_stats
|
||||
from util import device_stats
|
||||
from vertex_vision_model_garden_peft.train.vmg import callbacks
|
||||
from vertex_vision_model_garden_peft.train.vmg import eval_lib
|
||||
from vertex_vision_model_garden_peft.train.vmg import utils
|
||||
@@ -231,11 +232,6 @@ _PER_DEVICE_EVAL_BATCH_SIZE = flags.DEFINE_integer(
|
||||
'The per device batch size for model evaluation.',
|
||||
)
|
||||
|
||||
_EVAL_NUM_FEWSHOT = flags.DEFINE_integer(
|
||||
'eval_num_fewshot',
|
||||
None,
|
||||
'Run N-shot language model evaluation. Not implemented in `builtin_eval`.',
|
||||
)
|
||||
|
||||
_EVAL_LIMIT = flags.DEFINE_float(
|
||||
'eval_limit',
|
||||
@@ -255,8 +251,7 @@ _EVAL_METRIC_NAME = flags.DEFINE_list(
|
||||
_EVAL_DATASET = flags.DEFINE_string(
|
||||
'eval_dataset',
|
||||
None,
|
||||
'Overrides the default evaluation dataset path. In `builtin_eval` mode,'
|
||||
' this can be any Hugging Face dataset name or path.',
|
||||
'The Hugging Face dataset name or path to use for evaluation.',
|
||||
)
|
||||
|
||||
# We set the default eval split as `test`, based on observation from
|
||||
@@ -264,13 +259,13 @@ _EVAL_DATASET = flags.DEFINE_string(
|
||||
_EVAL_SPLIT = flags.DEFINE_string(
|
||||
'eval_split',
|
||||
'test',
|
||||
'Eval split name in the eval dataset for `builtin_eval`.',
|
||||
'Eval split name in the eval dataset.',
|
||||
)
|
||||
|
||||
_EVAL_TEMPLATE = flags.DEFINE_string(
|
||||
'eval_template',
|
||||
None,
|
||||
'Template for formatting language model evaluation data for `builtin_eval`.'
|
||||
'Template for formatting language model evaluation data.'
|
||||
' Must be a filename under `templates` folder, without `.json` extension,'
|
||||
' e.g. `alpaca`, or a Cloud Storage URI to a JSON file.',
|
||||
)
|
||||
@@ -278,7 +273,7 @@ _EVAL_TEMPLATE = flags.DEFINE_string(
|
||||
_EVAL_COLUMN = flags.DEFINE_string(
|
||||
'eval_column',
|
||||
None,
|
||||
'Eval column name in the eval dataset for `builtin_eval`.',
|
||||
'Eval column name in the eval dataset.',
|
||||
)
|
||||
|
||||
_METRIC_FOR_BEST_MODEL = flags.DEFINE_string(
|
||||
@@ -576,8 +571,8 @@ def finetune_instruct(
|
||||
"""Finetunes instruct."""
|
||||
logging.info(
|
||||
'on entering instruct_lora, %s,\n%s',
|
||||
utils.gpu_stats_str(),
|
||||
utils.cpu_stats_str(),
|
||||
device_stats.gpu_stats_str(),
|
||||
device_stats.cpu_stats_str(),
|
||||
)
|
||||
gradient_checkpointing_kwargs = {}
|
||||
# DDP provides limited support with the reentrant variant of gradient
|
||||
@@ -594,7 +589,7 @@ def finetune_instruct(
|
||||
access_token=access_token,
|
||||
)
|
||||
|
||||
train_dataset_with_template = (
|
||||
train_dataset, train_dataset_with_template = (
|
||||
dataset_validation_util.load_dataset_with_template(
|
||||
train_dataset,
|
||||
split=train_split,
|
||||
@@ -621,18 +616,20 @@ def finetune_instruct(
|
||||
'getting tuning data stats with effective batch size %s',
|
||||
effective_batch_size,
|
||||
)
|
||||
train_dataset_stats = utils.get_dataset_stats(
|
||||
train_dataset_with_template,
|
||||
tokenizer,
|
||||
train_column,
|
||||
effective_batch_size,
|
||||
train_dataset_stats = dataset_stats.get_dataset_stats(
|
||||
raw=train_dataset,
|
||||
templated=train_dataset_with_template,
|
||||
template=train_template,
|
||||
tokenizer=tokenizer,
|
||||
column=train_column,
|
||||
effective_batch_size=effective_batch_size,
|
||||
)
|
||||
logging.info('stats: %s', train_dataset_stats)
|
||||
tuning_data_stats_file = dataset_validation_util.force_gcs_fuse_path(
|
||||
tuning_data_stats_file
|
||||
)
|
||||
with open(tuning_data_stats_file, 'w') as out_f:
|
||||
json.dump(dataclasses.asdict(train_dataset_stats), out_f)
|
||||
json.dump(train_dataset_stats, out_f)
|
||||
|
||||
model = utils.load_model(
|
||||
pretrained_model_name_or_path=pretrained_model_name_or_path,
|
||||
@@ -663,7 +660,9 @@ def finetune_instruct(
|
||||
# `get_peft_model`, which may revert other changes we did before. That's why
|
||||
# we are calling `get_peft_model` explicitly here.
|
||||
model = get_peft_model(model, peft_config)
|
||||
|
||||
adapter_for_eval_dir = os.path.join(output_dir, 'adapter_for_eval')
|
||||
logging.info('saving adapter for evaluation to %s...', adapter_for_eval_dir)
|
||||
peft_config.save_pretrained(adapter_for_eval_dir)
|
||||
# This is to work-around mix-precision training. This issue is not fixed as
|
||||
# of transformers==4.41.2.
|
||||
# See b/332760883#comment30 for more details.
|
||||
@@ -840,7 +839,6 @@ def main(unused_argv: Sequence[str]) -> None:
|
||||
if _EVAL_DATASET.value:
|
||||
eval_config = eval_lib.EvalConfig(
|
||||
per_device_batch_size=_PER_DEVICE_EVAL_BATCH_SIZE.value,
|
||||
num_fewshot=_EVAL_NUM_FEWSHOT.value,
|
||||
limit=_EVAL_LIMIT.value,
|
||||
metric_name=_EVAL_METRIC_NAME.value,
|
||||
steps=_EVAL_STEPS.value,
|
||||
|
||||
+11
-2
@@ -31,7 +31,7 @@ _MERGE_BASE_AND_LORA_OUTPUT_DIR = flags.DEFINE_string(
|
||||
|
||||
_MERGE_MODEL_PRECISION_MODE = flags.DEFINE_enum(
|
||||
'merge_model_precision_mode',
|
||||
constants.PRECISION_MODE_16,
|
||||
constants.PRECISION_MODE_16B,
|
||||
[
|
||||
constants.PRECISION_MODE_4,
|
||||
constants.PRECISION_MODE_8,
|
||||
@@ -86,10 +86,19 @@ def main(unused_argv: Sequence[str]) -> None:
|
||||
)
|
||||
)
|
||||
|
||||
finetuned_lora_model_dir = fileutils.force_gcs_path(
|
||||
_FINETUNED_LORA_MODEL_DIR.value
|
||||
)
|
||||
if dataset_validation_util.is_gcs_path(finetuned_lora_model_dir):
|
||||
finetuned_lora_model_dir = (
|
||||
dataset_validation_util.download_gcs_uri_to_local(
|
||||
finetuned_lora_model_dir
|
||||
)
|
||||
)
|
||||
utils.merge_causal_language_model_with_lora(
|
||||
pretrained_model_name_or_path=pretrained_model_name_or_path,
|
||||
precision_mode=_MERGE_MODEL_PRECISION_MODE.value,
|
||||
finetuned_lora_model_dir=_FINETUNED_LORA_MODEL_DIR.value,
|
||||
finetuned_lora_model_dir=finetuned_lora_model_dir,
|
||||
merged_model_output_dir=_MERGE_BASE_AND_LORA_OUTPUT_DIR.value,
|
||||
access_token=_HUGGINGFACE_ACCESS_TOKEN.value,
|
||||
)
|
||||
|
||||
-232
@@ -1,232 +0,0 @@
|
||||
"""Sequence classification with LoRA models."""
|
||||
|
||||
from typing import Sequence
|
||||
|
||||
from absl import app
|
||||
from absl import flags
|
||||
from datasets import load_dataset
|
||||
import evaluate
|
||||
from peft import get_peft_model
|
||||
from peft import LoraConfig
|
||||
import torch
|
||||
from torch.optim import AdamW
|
||||
from torch.utils.data import DataLoader
|
||||
from tqdm import tqdm
|
||||
from transformers import AutoModelForSequenceClassification
|
||||
from transformers import AutoTokenizer
|
||||
from transformers import get_linear_schedule_with_warmup
|
||||
|
||||
from util import dataset_validation_util
|
||||
|
||||
|
||||
_PRETRAINED_MODEL_NAME_OR_PATH = flags.DEFINE_string(
|
||||
"pretrained_model_name_or_path",
|
||||
None,
|
||||
"The pretrained model name or path. Supported models can be causal language"
|
||||
" modeling models from https://github.com/huggingface/peft/tree/main. Note,"
|
||||
" there might be different paddings for different models. This tool assumes"
|
||||
" the pretrained_model_name_or_path contains model name, and then choose"
|
||||
" proper padding methods. e.g. it must contain `llama` for `Llama2"
|
||||
" models`.",
|
||||
)
|
||||
|
||||
_OUTPUT_DIR = flags.DEFINE_string(
|
||||
"output_dir",
|
||||
None,
|
||||
"The output directory.",
|
||||
)
|
||||
|
||||
_DATASET_NAME = flags.DEFINE_string(
|
||||
"dataset_name",
|
||||
None,
|
||||
"The dataset name in huggingface.",
|
||||
)
|
||||
|
||||
_LORA_RANK = flags.DEFINE_integer(
|
||||
"lora_rank",
|
||||
16,
|
||||
"The rank of the update matrices, expressed in int. Lower rank results in"
|
||||
" smaller update matrices with fewer trainable parameters, referring to"
|
||||
" https://huggingface.co/docs/peft/conceptual_guides/lora.",
|
||||
)
|
||||
|
||||
_LORA_ALPHA = flags.DEFINE_integer(
|
||||
"lora_alpha",
|
||||
32,
|
||||
"LoRA scaling factor, referring to"
|
||||
" https://huggingface.co/docs/peft/conceptual_guides/lora.",
|
||||
)
|
||||
|
||||
_LORA_DROPOUT = flags.DEFINE_float(
|
||||
"lora_dropout",
|
||||
0.05,
|
||||
"dropout probability of the LoRA layers, referring to"
|
||||
" https://huggingface.co/docs/peft/task_guides/token-classification-lora.",
|
||||
)
|
||||
|
||||
_NUM_TRAIN_EPOCHS = flags.DEFINE_integer(
|
||||
"num_train_epochs",
|
||||
None,
|
||||
"The number of training epochs.",
|
||||
)
|
||||
|
||||
_BATCH_SIZE = flags.DEFINE_integer(
|
||||
"batch_size",
|
||||
32,
|
||||
"The batch size.",
|
||||
)
|
||||
|
||||
_LEARNING_RATE = flags.DEFINE_float(
|
||||
"learning_rate",
|
||||
2e-4,
|
||||
"The learning rate after the potential warmup period.",
|
||||
)
|
||||
|
||||
|
||||
def finetune_sequence_classification(
|
||||
pretrained_model_name_or_path: str,
|
||||
dataset_name: str,
|
||||
output_dir: str,
|
||||
lora_rank: int = 8,
|
||||
lora_alpha: int = 16,
|
||||
lora_dropout: float = 0.1,
|
||||
num_train_epochs: int = 20,
|
||||
batch_size: int = 32,
|
||||
learning_rate: float = 3e-4,
|
||||
) -> None:
|
||||
"""Finetunes sequence classification."""
|
||||
task = "mrpc"
|
||||
device = "cuda"
|
||||
|
||||
peft_config = LoraConfig(
|
||||
task_type="SEQ_CLS",
|
||||
inference_mode=False,
|
||||
r=lora_rank,
|
||||
lora_alpha=lora_alpha,
|
||||
lora_dropout=lora_dropout,
|
||||
)
|
||||
if any(k in pretrained_model_name_or_path for k in ("gpt", "opt", "bloom")):
|
||||
padding_side = "left"
|
||||
else:
|
||||
padding_side = "right"
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
pretrained_model_name_or_path, padding_side=padding_side
|
||||
)
|
||||
if getattr(tokenizer, "pad_token_id") is None:
|
||||
tokenizer.pad_token_id = tokenizer.eos_token_id
|
||||
|
||||
datasets = load_dataset(dataset_name, task)
|
||||
metric = evaluate.load(dataset_name, task)
|
||||
|
||||
def tokenize_function(examples):
|
||||
# max_length=None => use the model max length (it's actually the default)
|
||||
outputs = tokenizer(
|
||||
examples["sentence1"],
|
||||
examples["sentence2"],
|
||||
truncation=True,
|
||||
max_length=None,
|
||||
)
|
||||
return outputs
|
||||
|
||||
tokenized_datasets = datasets.map(
|
||||
tokenize_function,
|
||||
batched=True,
|
||||
remove_columns=["idx", "sentence1", "sentence2"],
|
||||
)
|
||||
|
||||
# We also rename the 'label' column to 'labels' which is the expected name for
|
||||
# labels by the models of the transformers library.
|
||||
tokenized_datasets = tokenized_datasets.rename_column("label", "labels")
|
||||
|
||||
def collate_fn(examples):
|
||||
return tokenizer.pad(examples, padding="longest", return_tensors="pt")
|
||||
|
||||
# Instantiate dataloaders.
|
||||
train_dataloader = DataLoader(
|
||||
tokenized_datasets["train"],
|
||||
shuffle=True,
|
||||
collate_fn=collate_fn,
|
||||
batch_size=batch_size,
|
||||
)
|
||||
eval_dataloader = DataLoader(
|
||||
tokenized_datasets["validation"],
|
||||
shuffle=False,
|
||||
collate_fn=collate_fn,
|
||||
batch_size=batch_size,
|
||||
)
|
||||
|
||||
model = AutoModelForSequenceClassification.from_pretrained(
|
||||
pretrained_model_name_or_path, return_dict=True
|
||||
)
|
||||
model = get_peft_model(model, peft_config)
|
||||
model.print_trainable_parameters()
|
||||
|
||||
optimizer = AdamW(params=model.parameters(), lr=learning_rate)
|
||||
|
||||
# Instantiate scheduler
|
||||
lr_scheduler = get_linear_schedule_with_warmup(
|
||||
optimizer=optimizer,
|
||||
num_warmup_steps=0.06 * (len(train_dataloader) * num_train_epochs),
|
||||
num_training_steps=(len(train_dataloader) * num_train_epochs),
|
||||
)
|
||||
|
||||
model.to(device)
|
||||
for epoch in range(num_train_epochs):
|
||||
model.train()
|
||||
for _, batch in enumerate(tqdm(train_dataloader)):
|
||||
batch.to(device)
|
||||
outputs = model(**batch)
|
||||
loss = outputs.loss
|
||||
loss.backward()
|
||||
optimizer.step()
|
||||
lr_scheduler.step()
|
||||
optimizer.zero_grad()
|
||||
|
||||
model.eval()
|
||||
for _, batch in enumerate(tqdm(eval_dataloader)):
|
||||
batch.to(device)
|
||||
with torch.no_grad():
|
||||
outputs = model(**batch)
|
||||
predictions = outputs.logits.argmax(dim=-1)
|
||||
references = batch["labels"]
|
||||
metric.add_batch(
|
||||
predictions=predictions,
|
||||
references=references,
|
||||
)
|
||||
|
||||
eval_metric = metric.compute()
|
||||
print(f"epoch {epoch}:", eval_metric)
|
||||
|
||||
model.save_pretrained(output_dir)
|
||||
|
||||
|
||||
def main(unused_argv: Sequence[str]) -> None:
|
||||
if dataset_validation_util.is_gcs_path(_PRETRAINED_MODEL_NAME_OR_PATH.value):
|
||||
pretrained_model_name_or_path = (
|
||||
dataset_validation_util.download_gcs_uri_to_local(
|
||||
_PRETRAINED_MODEL_NAME_OR_PATH.value
|
||||
)
|
||||
)
|
||||
else:
|
||||
pretrained_model_name_or_path = _PRETRAINED_MODEL_NAME_OR_PATH.value
|
||||
pretrained_model_path = dataset_validation_util.force_gcs_fuse_path(
|
||||
pretrained_model_name_or_path
|
||||
)
|
||||
output_dir = dataset_validation_util.force_gcs_fuse_path(_OUTPUT_DIR.value)
|
||||
|
||||
finetune_sequence_classification(
|
||||
pretrained_model_name_or_path=pretrained_model_path,
|
||||
dataset_name=_DATASET_NAME.value,
|
||||
output_dir=output_dir,
|
||||
lora_rank=_LORA_RANK.value,
|
||||
lora_alpha=_LORA_ALPHA.value,
|
||||
lora_dropout=_LORA_DROPOUT.value,
|
||||
num_train_epochs=int(_NUM_TRAIN_EPOCHS.value),
|
||||
batch_size=_BATCH_SIZE.value,
|
||||
learning_rate=_LEARNING_RATE.value,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app.run(main)
|
||||
@@ -0,0 +1,7 @@
|
||||
{
|
||||
"description": "Chat template used by Qwen 2.5.",
|
||||
"source": "https://huggingface.co/Qwen/Qwen2.5-72B-Instruct/blob/main/tokenizer_config.json#L198",
|
||||
"chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0]['role'] == 'system' %}\n {{- messages[0]['content'] }}\n {%- else %}\n {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n {%- endif %}\n {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0]['role'] == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n {%- else %}\n {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {{- '<|im_start|>' + message.role }}\n {%- if message.content %}\n {{- '\\n' + message.content }}\n {%- endif %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '\\n<tool_call>\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {{- tool_call.arguments | tojson }}\n {{- '}\\n</tool_call>' }}\n {%- endfor %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- message.content }}\n {{- '\\n</tool_response>' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n",
|
||||
"instruction_separator": "<|im_start|>user\n",
|
||||
"response_separator": "<|im_start|>assistant\n"
|
||||
}
|
||||
+17
-4
@@ -64,6 +64,7 @@ class TrainerThroughputTest(test_util.TestBase):
|
||||
'Mistral-7B-v0.1',
|
||||
'Mixtral-8x7B-v0.1',
|
||||
'gemma-2-9b-it',
|
||||
'Qwen2.5-32B-Instruct',
|
||||
],
|
||||
precision=['4bit', '8bit', 'bfloat16'],
|
||||
max_seq_length=list(range(4 * 1024, 24 * 1024 + 1, 4 * 1024)),
|
||||
@@ -89,6 +90,7 @@ class TrainerThroughputTest(test_util.TestBase):
|
||||
'Mistral-7B-v0.1',
|
||||
'Mixtral-8x7B-v0.1',
|
||||
'gemma-2-9b-it',
|
||||
'Qwen2.5-32B-Instruct',
|
||||
],
|
||||
precision=['4bit', '8bit', 'bfloat16'],
|
||||
max_seq_length=list(range(4 * 1024, 24 * 1024 + 1, 4 * 1024)),
|
||||
@@ -119,7 +121,11 @@ class TrainerThroughputTest(test_util.TestBase):
|
||||
self.assertEqual(self.run_cmd_and_handle_failure(), 0)
|
||||
|
||||
@parameterized.product(
|
||||
model_name=['llama3.1-8b-hf', 'llama3.1-70b-hf'],
|
||||
model_name=[
|
||||
'llama3.1-8b-hf',
|
||||
'llama3.1-70b-hf',
|
||||
'Qwen2.5-32B-Instruct',
|
||||
],
|
||||
precision=['4bit', '8bit', 'bfloat16'],
|
||||
max_seq_length=list(range(4 * 1024, 24 * 1024 + 1, 4 * 1024)),
|
||||
num_gpus=[8],
|
||||
@@ -136,9 +142,16 @@ class TrainerThroughputTest(test_util.TestBase):
|
||||
self.test_suite_output_dir,
|
||||
f'bm_fsdp_{num_gpus}gpu_{model_name}_{precision}.txt',
|
||||
)
|
||||
self.task_cmd_builder.config_file = (
|
||||
'vertex_vision_model_garden_peft/llama_fsdp_8gpu.yaml'
|
||||
)
|
||||
if 'llama' in model_name.lower():
|
||||
self.task_cmd_builder.config_file = (
|
||||
'vertex_vision_model_garden_peft/llama2_fsdp_8gpu.yaml'
|
||||
)
|
||||
elif 'qwen' in model_name.lower():
|
||||
self.task_cmd_builder.config_file = (
|
||||
'vertex_vision_model_garden_peft/qwen2_fsdp_8gpu.yaml'
|
||||
)
|
||||
else:
|
||||
self.fail(f'Unsupported model: {model_name}')
|
||||
|
||||
self.command_builder.add_env_var(
|
||||
'CUDA_VISIBLE_DEVICES', ','.join([str(x) for x in range(0, num_gpus)])
|
||||
|
||||
+64
@@ -140,6 +140,70 @@ class TrainedModelQualityTest(test_util.TestBase):
|
||||
|
||||
self.assertEqual(self.run_cmd(), 0)
|
||||
|
||||
@parameterized.named_parameters(
|
||||
('Qwen2.5-32B-Instruct', 'Qwen2.5-32B-Instruct'),
|
||||
)
|
||||
def test_qwen_model_deepspeed(self, model_name):
|
||||
self.setup_output_dir(f'test_deepspeed_{model_name}')
|
||||
self.task_cmd_builder.pretrained_model_name_or_path = model_name
|
||||
self.task_cmd_builder.config_file = (
|
||||
'vertex_vision_model_garden_peft/deepspeed_zero2_8gpu.yaml'
|
||||
)
|
||||
self.task_cmd_builder.train_dataset = test_util.get_test_data_path(
|
||||
'llama-tuning-test/opposite-examples-train.jsonl'
|
||||
)
|
||||
self.task_cmd_builder.train_split = 'train'
|
||||
self.task_cmd_builder.train_column = 'messages'
|
||||
self.task_cmd_builder.train_template = 'qwen2_5'
|
||||
self.task_cmd_builder.eval_dataset = test_util.get_test_data_path(
|
||||
'llama-tuning-test/opposite-examples-eval.jsonl'
|
||||
)
|
||||
self.task_cmd_builder.eval_split = 'train'
|
||||
self.task_cmd_builder.eval_column = self.task_cmd_builder.train_column
|
||||
self.task_cmd_builder.eval_template = self.task_cmd_builder.train_template
|
||||
|
||||
self.command_builder.add_env_var('CUDA_VISIBLE_DEVICES', '0,1,2,3,4,5,6,7')
|
||||
# Note(lavrai): The following parameters are needed for the opposite-word
|
||||
# dataset to converge properly.
|
||||
self.task_cmd_builder.gradient_accumulation_steps = 1
|
||||
self.task_cmd_builder.num_train_epochs = 10.0
|
||||
self.task_cmd_builder.logging_steps = 1
|
||||
|
||||
self.assertEqual(self.run_cmd(), 0)
|
||||
|
||||
@parameterized.named_parameters(
|
||||
('Qwen2.5-32B-Instruct', 'Qwen2.5-32B-Instruct'),
|
||||
)
|
||||
def test_qwen_model_fsdp(self, model_name):
|
||||
self.setup_output_dir(f'test_fsdp_{model_name}')
|
||||
self.task_cmd_builder.pretrained_model_name_or_path = (
|
||||
test_util.get_pretrained_model_name_or_path(model_name)
|
||||
)
|
||||
self.task_cmd_builder.config_file = (
|
||||
'vertex_vision_model_garden_peft/qwen2_fsdp_8gpu.yaml'
|
||||
)
|
||||
self.task_cmd_builder.train_dataset = test_util.get_test_data_path(
|
||||
'llama-tuning-test/opposite-examples-train.jsonl'
|
||||
)
|
||||
self.task_cmd_builder.train_split = 'train'
|
||||
self.task_cmd_builder.train_column = 'messages'
|
||||
self.task_cmd_builder.train_template = 'qwen2_5'
|
||||
self.task_cmd_builder.eval_dataset = test_util.get_test_data_path(
|
||||
'llama-tuning-test/opposite-examples-eval.jsonl'
|
||||
)
|
||||
self.task_cmd_builder.eval_split = 'train'
|
||||
self.task_cmd_builder.eval_column = self.task_cmd_builder.train_column
|
||||
self.task_cmd_builder.eval_template = self.task_cmd_builder.train_template
|
||||
|
||||
self.command_builder.add_env_var('CUDA_VISIBLE_DEVICES', '0,1,2,3,4,5,6,7')
|
||||
# Note(lavrai): The following parameters are needed for the opposite-word
|
||||
# dataset to converge properly.
|
||||
self.task_cmd_builder.gradient_accumulation_steps = 1
|
||||
self.task_cmd_builder.num_train_epochs = 10.0
|
||||
self.task_cmd_builder.logging_steps = 1
|
||||
|
||||
self.assertEqual(self.run_cmd(), 0)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
absltest.main()
|
||||
|
||||
+57
-134
@@ -9,19 +9,17 @@ environment. Otherwise, `python3` is used.
|
||||
|
||||
import argparse
|
||||
from collections.abc import MutableSequence, Sequence
|
||||
import json
|
||||
import multiprocessing
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from absl import app
|
||||
from absl import flags
|
||||
from absl import logging
|
||||
from util import dataset_validation_util
|
||||
from vertex_vision_model_garden_peft.train.vmg import gcs_syncer
|
||||
from util import cluster_spec
|
||||
from vertex_vision_model_garden_peft.train.vmg import utils
|
||||
from util import constants
|
||||
from util import fileutils
|
||||
from util import gcs_syncer
|
||||
from util import hypertune_utils
|
||||
|
||||
|
||||
@@ -41,9 +39,12 @@ _TASK_TO_SCRIPT = {
|
||||
constants.INSTRUCT_LORA: (
|
||||
'vertex_vision_model_garden_peft/train/vmg/instruct_lora.py'
|
||||
),
|
||||
constants.MERGE_CAUSAL_LANGUAGE_MODEL_LORA: 'vertex_vision_model_garden_peft/train/vmg/merge_causal_language_model_lora.py',
|
||||
constants.SEQUENCE_CLASSIFICATION_LORA: 'vertex_vision_model_garden_peft/train/vmg/sequence_classification_lora.py',
|
||||
constants.VALIDATE_DATASET_WITH_TEMPLATE: 'vertex_vision_model_garden_peft/train/vmg/validate_dataset_with_template.py',
|
||||
constants.MERGE_CAUSAL_LANGUAGE_MODEL_LORA: (
|
||||
'vertex_vision_model_garden_peft/train/vmg/merge_causal_language_model_lora.py'
|
||||
),
|
||||
constants.VALIDATE_DATASET_WITH_TEMPLATE: (
|
||||
'vertex_vision_model_garden_peft/train/vmg/validate_dataset_with_template.py'
|
||||
),
|
||||
constants.RUN_TESTS: 'vertex_vision_model_garden_peft/tests/run_tests.py',
|
||||
}
|
||||
|
||||
@@ -71,53 +72,17 @@ def launch_script_cmd(
|
||||
|
||||
def _get_accelerate_args() -> argparse.Namespace:
|
||||
"""Returns the accelerate args."""
|
||||
# For the format of the cluster spec, see
|
||||
# https://cloud.google.com/vertex-ai/docs/training/distributed-training#cluster-spec-format # pylint: disable=line-too-long
|
||||
cluster_spec = os.getenv('CLUSTER_SPEC', default=None)
|
||||
if not cluster_spec:
|
||||
return argparse.Namespace()
|
||||
logging.info('CLUSTER_SPEC: %s', cluster_spec)
|
||||
|
||||
cluster_data = json.loads(cluster_spec)
|
||||
if (
|
||||
'workerpool1' not in cluster_data['cluster']
|
||||
or not cluster_data['cluster']['workerpool1']
|
||||
):
|
||||
return argparse.Namespace()
|
||||
|
||||
# Get primary node info
|
||||
primary_node = cluster_data['cluster']['workerpool0'][0]
|
||||
logging.info('primary node: %s', primary_node)
|
||||
primary_node_addr, primary_node_port = primary_node.split(':')
|
||||
logging.info('primary node address: %s', primary_node_addr)
|
||||
logging.info('primary node port: %s', primary_node_port)
|
||||
|
||||
# Determine node rank of this machine
|
||||
workerpool = cluster_data['task']['type']
|
||||
if workerpool == 'workerpool0':
|
||||
node_rank = 0
|
||||
elif workerpool == 'workerpool1':
|
||||
# Add 1 for the primary node, since `index` is the index of workerpool1.
|
||||
node_rank = cluster_data['task']['index'] + 1
|
||||
else:
|
||||
raise ValueError(
|
||||
'Only workerpool0 and workerpool1 are supported. Unknown workerpool:'
|
||||
f' {workerpool}'
|
||||
)
|
||||
logging.info('node rank: %s', node_rank)
|
||||
|
||||
# Calculate total nodes
|
||||
num_worker_nodes = len(cluster_data['cluster']['workerpool1'])
|
||||
num_nodes = num_worker_nodes + 1 # Add 1 for the primary node
|
||||
logging.info('num nodes: %s', num_nodes)
|
||||
|
||||
primary_node_addr, primary_node_port, node_rank, num_nodes = (
|
||||
cluster_spec.get_cluster_spec()
|
||||
)
|
||||
accelerate_args = argparse.Namespace()
|
||||
accelerate_args.machine_rank = node_rank
|
||||
accelerate_args.num_machines = num_nodes
|
||||
accelerate_args.main_process_ip = primary_node_addr
|
||||
accelerate_args.main_process_port = primary_node_port
|
||||
accelerate_args.max_restarts = 0
|
||||
accelerate_args.monitor_interval = 120
|
||||
if num_nodes > 1:
|
||||
accelerate_args.machine_rank = node_rank
|
||||
accelerate_args.num_machines = num_nodes
|
||||
accelerate_args.main_process_ip = primary_node_addr
|
||||
accelerate_args.main_process_port = primary_node_port
|
||||
accelerate_args.max_restarts = 0
|
||||
accelerate_args.monitor_interval = 120
|
||||
|
||||
return accelerate_args
|
||||
|
||||
@@ -131,45 +96,6 @@ def _append_args_to_command_in_place(
|
||||
command.append(f'--{key}={value}')
|
||||
|
||||
|
||||
def _is_gcs_or_gcsfuse_path(path: str) -> bool:
|
||||
"""Returns if the path is a GCS or gcsfuse path.
|
||||
|
||||
Args:
|
||||
path: The path to check.
|
||||
|
||||
Returns:
|
||||
True if the path is a GCS or gcsfuse path.
|
||||
"""
|
||||
return path.startswith(
|
||||
(constants.GCS_URI_PREFIX, constants.GCSFUSE_URI_PREFIX)
|
||||
)
|
||||
|
||||
|
||||
def _manage_training_path(path: str, node_rank: int) -> tuple[str, str]:
|
||||
"""Returns local dir and GCS location for the given path if the given path is a GCS or gcsfuse path.
|
||||
|
||||
It will also create a local directory if it does not exist. Othereise, it
|
||||
returns the same path.
|
||||
|
||||
Args:
|
||||
path: The local or GCS path to manage.
|
||||
node_rank: The node rank to be appended to the GCS path.
|
||||
|
||||
Returns:
|
||||
The local and GCS paths.
|
||||
"""
|
||||
local_dir = path
|
||||
gcs_dir = path
|
||||
if _is_gcs_or_gcsfuse_path(path):
|
||||
local_dir = os.path.join(
|
||||
constants.LOCAL_OUTPUT_DIR,
|
||||
dataset_validation_util.force_gcs_fuse_path(path)[1:],
|
||||
)
|
||||
gcs_dir = fileutils.force_gcs_path(path)
|
||||
os.makedirs(local_dir, exist_ok=True)
|
||||
return local_dir, os.path.join(gcs_dir, f'node-{node_rank}')
|
||||
|
||||
|
||||
def _get_train_and_maybe_merge_cmd_and_dirs_to_sync(
|
||||
task_type: str, config_file: str, unknown: Sequence[str]
|
||||
) -> Sequence[Sequence[str]]:
|
||||
@@ -203,11 +129,11 @@ def _get_train_and_maybe_merge_cmd_and_dirs_to_sync(
|
||||
dataset_validation_util.force_gcs_fuse_path(training_args.output_dir)
|
||||
)
|
||||
|
||||
local_output_dir, gcs_output_dir = _manage_training_path(
|
||||
local_output_dir, gcs_output_dir = gcs_syncer.manage_sync_path(
|
||||
training_args.output_dir, node_rank
|
||||
)
|
||||
training_args.output_dir = local_output_dir
|
||||
if _is_gcs_or_gcsfuse_path(gcs_output_dir):
|
||||
if gcs_syncer.is_gcs_or_gcsfuse_path(gcs_output_dir):
|
||||
dirs_to_sync.append((local_output_dir, gcs_output_dir))
|
||||
|
||||
# Merge only flags.
|
||||
@@ -217,11 +143,11 @@ def _get_train_and_maybe_merge_cmd_and_dirs_to_sync(
|
||||
merge_args, unknown = merge_parser.parse_known_args(unknown)
|
||||
|
||||
if merge_args.merge_base_and_lora_output_dir:
|
||||
merge_local_dir, merge_gcs_dir = _manage_training_path(
|
||||
merge_args.merge_base_and_lora_output_dir, node_rank
|
||||
merge_local_dir, merge_gcs_dir = gcs_syncer.manage_sync_path(
|
||||
merge_args.merge_base_and_lora_output_dir, None
|
||||
)
|
||||
merge_args.merge_base_and_lora_output_dir = merge_local_dir
|
||||
if _is_gcs_or_gcsfuse_path(merge_gcs_dir):
|
||||
if gcs_syncer.is_gcs_or_gcsfuse_path(merge_gcs_dir):
|
||||
dirs_to_sync.append((merge_local_dir, merge_gcs_dir))
|
||||
|
||||
# Common flags shared by merging and training.
|
||||
@@ -239,8 +165,10 @@ def _get_train_and_maybe_merge_cmd_and_dirs_to_sync(
|
||||
# Only the main node runs merging.
|
||||
if merge_args.merge_base_and_lora_output_dir and node_rank == 0:
|
||||
lora_dir = utils.get_final_checkpoint_path(training_args.output_dir)
|
||||
lora_local_dir, lora_gcs_dir = _manage_training_path(lora_dir, node_rank)
|
||||
if _is_gcs_or_gcsfuse_path(lora_gcs_dir):
|
||||
lora_local_dir, lora_gcs_dir = gcs_syncer.manage_sync_path(
|
||||
lora_dir, node_rank
|
||||
)
|
||||
if gcs_syncer.is_gcs_or_gcsfuse_path(lora_gcs_dir):
|
||||
dirs_to_sync.append((lora_local_dir, lora_gcs_dir))
|
||||
|
||||
merge_cmd = [
|
||||
@@ -263,46 +191,37 @@ def _get_train_and_maybe_merge_cmd_and_dirs_to_sync(
|
||||
return commands, dirs_to_sync
|
||||
|
||||
|
||||
def _setup_gcs_rsync(
|
||||
dirs_to_sync: Sequence[tuple[str, str]],
|
||||
mp_queue: multiprocessing.Queue,
|
||||
gcs_rsync_interval_secs: int,
|
||||
) -> multiprocessing.Process:
|
||||
"""Sets up the GCS rsync process.
|
||||
def _get_merge_cmd_and_dirs_to_sync(
|
||||
task_type: str, config_file: str, unknown: Sequence[str]
|
||||
) -> Sequence[Sequence[str]]:
|
||||
"""Returns the merge command and dirs to sync.
|
||||
|
||||
Args:
|
||||
dirs_to_sync: The absolute directory paths which will be synced to GCS.
|
||||
mp_queue: The multiprocessing queue to check if the training is finished.
|
||||
gcs_rsync_interval_secs: Integer, interval in seconds to run gcs rsync.
|
||||
task_type: The task type.
|
||||
config_file: The accelerate config file path.
|
||||
unknown: The unknown args which are not recognised by the parser.
|
||||
|
||||
Returns:
|
||||
The GCS rsync process.
|
||||
The bash commands to execute and the directories to sync.
|
||||
"""
|
||||
rsync_process = multiprocessing.Process(
|
||||
target=gcs_syncer.start_gcs_rsync,
|
||||
args=(dirs_to_sync, mp_queue, gcs_rsync_interval_secs),
|
||||
)
|
||||
rsync_process.start()
|
||||
return rsync_process
|
||||
# Merge only flags.
|
||||
merge_parser = argparse.ArgumentParser()
|
||||
merge_parser.add_argument('--merge_base_and_lora_output_dir')
|
||||
merge_args, unknown = merge_parser.parse_known_args(unknown)
|
||||
|
||||
|
||||
def _cleanup_gcs_rsync(
|
||||
rsync_process: multiprocessing.Process, mp_queue: multiprocessing.Queue
|
||||
) -> None:
|
||||
"""Cleans up the GCS rsync process.
|
||||
|
||||
Args:
|
||||
rsync_process: The GCS rsync process.
|
||||
mp_queue: The multiprocessing queue.
|
||||
"""
|
||||
mp_queue.put('training finished')
|
||||
rsync_process.join()
|
||||
if rsync_process.exitcode == 0:
|
||||
logging.info('Artifacts have been uploaded to GCS.')
|
||||
else:
|
||||
logging.error(
|
||||
'GCS rsync process failed with exit code %d.', rsync_process.exitcode
|
||||
dirs_to_sync = []
|
||||
if merge_args.merge_base_and_lora_output_dir:
|
||||
merge_local_dir, merge_gcs_dir = gcs_syncer.manage_sync_path(
|
||||
merge_args.merge_base_and_lora_output_dir, None
|
||||
)
|
||||
merge_args.merge_base_and_lora_output_dir = merge_local_dir
|
||||
if gcs_syncer.is_gcs_or_gcsfuse_path(merge_gcs_dir):
|
||||
dirs_to_sync.append((merge_local_dir, merge_gcs_dir))
|
||||
|
||||
cmd = launch_script_cmd(_TASK_TO_SCRIPT[task_type], config_file)
|
||||
_append_args_to_command_in_place(merge_args, cmd)
|
||||
cmd.extend(unknown)
|
||||
return [cmd], dirs_to_sync
|
||||
|
||||
|
||||
def main(unused_argv: Sequence[str]) -> None:
|
||||
@@ -335,6 +254,10 @@ def main(unused_argv: Sequence[str]) -> None:
|
||||
commands, dirs_to_sync = _get_train_and_maybe_merge_cmd_and_dirs_to_sync(
|
||||
task_type=task, config_file=args.config_file, unknown=unknown
|
||||
)
|
||||
elif task in [constants.MERGE_CAUSAL_LANGUAGE_MODEL_LORA]:
|
||||
commands, dirs_to_sync = _get_merge_cmd_and_dirs_to_sync(
|
||||
task_type=task, config_file=args.config_file, unknown=unknown
|
||||
)
|
||||
else:
|
||||
assert task in _TASK_TO_SCRIPT
|
||||
cmd = launch_script_cmd(_TASK_TO_SCRIPT[task], args.config_file)
|
||||
@@ -344,7 +267,7 @@ def main(unused_argv: Sequence[str]) -> None:
|
||||
rsync_process = None
|
||||
mp_queue = multiprocessing.Queue(maxsize=1)
|
||||
if dirs_to_sync:
|
||||
rsync_process = _setup_gcs_rsync(
|
||||
rsync_process = gcs_syncer.setup_gcs_rsync(
|
||||
dirs_to_sync, mp_queue, args.gcs_rsync_interval_secs
|
||||
)
|
||||
|
||||
@@ -361,7 +284,7 @@ def main(unused_argv: Sequence[str]) -> None:
|
||||
rsync_process.terminate()
|
||||
raise e
|
||||
if rsync_process is not None:
|
||||
_cleanup_gcs_rsync(rsync_process, mp_queue)
|
||||
gcs_syncer.cleanup_gcs_rsync(rsync_process, mp_queue)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
"""Common libraries for PEFT."""
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
import dataclasses
|
||||
import datetime
|
||||
import gc
|
||||
import os
|
||||
@@ -11,12 +10,9 @@ from absl import logging
|
||||
import accelerate
|
||||
from accelerate import DistributedType
|
||||
from accelerate import PartialState
|
||||
import numpy as np
|
||||
import peft
|
||||
from peft import PeftModel
|
||||
from peft import prepare_model_for_kbit_training
|
||||
import psutil
|
||||
import pynvml
|
||||
import torch
|
||||
import transformers
|
||||
from transformers import AutoModelForCausalLM
|
||||
@@ -28,7 +24,6 @@ import trl
|
||||
from util import dataset_validation_util
|
||||
from util import constants
|
||||
|
||||
|
||||
_LLAMA_3_1_405B_MODEL_ID = "Meta-Llama-3.1-405B"
|
||||
_LOCAL_MERGED_MODEL_DIR = "/tmp/merged_model"
|
||||
_GEMMA2_MODEL = "gemma-2"
|
||||
@@ -126,7 +121,7 @@ def load_model(
|
||||
"device_map": device_map,
|
||||
"torch_dtype": torch_dtype,
|
||||
"quantization_config": quantization_config,
|
||||
"trust_remote_code": True,
|
||||
"trust_remote_code": False,
|
||||
"token": access_token,
|
||||
"attn_implementation": attn_implementation,
|
||||
}
|
||||
@@ -310,171 +305,12 @@ def convert_model_to_fp8(
|
||||
PartialState().wait_for_everyone()
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class TuningDataStats:
|
||||
tuning_dataset_example_count: int
|
||||
total_billable_token_count: int
|
||||
tuning_step_count: int
|
||||
|
||||
|
||||
def get_dataset_stats(
|
||||
dataset: Any,
|
||||
tokenizer: transformers.PreTrainedTokenizer,
|
||||
column: str,
|
||||
effective_batch_size: int,
|
||||
) -> TuningDataStats:
|
||||
"""Calculates dataset statistics, e.g., total number of tokens."""
|
||||
tokenized_dataset = dataset.map(lambda x: tokenizer(x[column]))
|
||||
inputs = tokenized_dataset["input_ids"]
|
||||
tuning_dataset_example_count = int(len(inputs))
|
||||
total_billable_token_count = int(np.sum([len(ex) for ex in inputs]))
|
||||
tuning_step_count = (
|
||||
tuning_dataset_example_count + effective_batch_size - 1
|
||||
) // effective_batch_size
|
||||
return TuningDataStats(
|
||||
tuning_dataset_example_count,
|
||||
total_billable_token_count,
|
||||
tuning_step_count,
|
||||
)
|
||||
|
||||
|
||||
def force_gc():
|
||||
"""Collects garbage immediately to release unused CPU/GPU resources."""
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class GpuStats:
|
||||
"""Holds information about GPU usage stats.
|
||||
|
||||
For memory related, see
|
||||
https://pytorch.org/docs/stable/notes/cuda.html#cuda-memory-management
|
||||
"""
|
||||
|
||||
# total memory
|
||||
total_mem: float
|
||||
# memory occupied.
|
||||
occupied: float
|
||||
# memory reserved, but not used.
|
||||
unused: float
|
||||
# nvidia-smi usually reports more memory usages than pytorch (for driver,
|
||||
# kernel and etc). `smi_diff` tracks this difference.
|
||||
smi_diff: float
|
||||
# Gpu utilization.
|
||||
util: float
|
||||
|
||||
# Allows unpacking operation like
|
||||
# total_mem, occupied, unused, smi_diff, util = GpuStats(...)
|
||||
# See https://stackoverflow.com/a/70753113
|
||||
def __iter__(self):
|
||||
return iter(dataclasses.astuple(self))
|
||||
|
||||
|
||||
def gpu_stats() -> GpuStats:
|
||||
"""Reports GPU memory usage and utilization."""
|
||||
# See https://pytorch.org/docs/stable/notes/cuda.html#memory-management
|
||||
bytes_per_gb = 1024.0**3
|
||||
device = torch.cuda.current_device()
|
||||
occupied = torch.cuda.memory_allocated(device) / bytes_per_gb
|
||||
reserved = torch.cuda.memory_reserved(device) / bytes_per_gb
|
||||
unused = reserved - occupied
|
||||
|
||||
def smi_mem(device):
|
||||
try:
|
||||
pynvml.nvmlInit()
|
||||
handle = pynvml.nvmlDeviceGetHandleByIndex(device)
|
||||
info = pynvml.nvmlDeviceGetMemoryInfo(handle)
|
||||
return info.used / bytes_per_gb
|
||||
except pynvml.NVMLError:
|
||||
return 0.0
|
||||
|
||||
mem_used_smi = smi_mem(device)
|
||||
smi_diff = mem_used_smi - reserved
|
||||
|
||||
util = torch.cuda.utilization(device)
|
||||
return GpuStats(mem_used_smi, occupied, unused, smi_diff, util)
|
||||
|
||||
|
||||
def gpu_stats_str(stats: GpuStats | None = None) -> str:
|
||||
if stats is None:
|
||||
stats = gpu_stats()
|
||||
total, occupied, unused, smi_diff, util = stats
|
||||
return (
|
||||
f"GPU memory: {total:.2f}({occupied=:.2f}, {unused=:.2f},"
|
||||
f" {smi_diff=:.2f}) GB. Utilization: {util:.2f}%"
|
||||
)
|
||||
|
||||
|
||||
@dataclasses.dataclass
|
||||
class CpuStats:
|
||||
"""Holds information about CPU usage stats."""
|
||||
|
||||
# Total CPU virtual memory i.e. virtual memory allocated + unallocated.
|
||||
total_virtual_mem: float
|
||||
# CPU virtual memory available for use.
|
||||
unallocated_virtual_mem: float
|
||||
# CPU virtual memory already used.
|
||||
allocated_virtual_mem: float
|
||||
# Total CPU swap memory i.e. swap memory allocated + unallocated.
|
||||
total_swap_mem: float
|
||||
# CPU swap memory available for use.
|
||||
unallocated_swap_mem: float
|
||||
# CPU swap memory already used.
|
||||
allocated_swap_mem: float
|
||||
# CPU utilization percentage.
|
||||
utilization: float
|
||||
|
||||
|
||||
def cpu_stats() -> CpuStats:
|
||||
"""Reports CPU memory usage and utilization."""
|
||||
|
||||
# https://psutil.readthedocs.io/en/latest/#memory
|
||||
gb = 1024.0**3
|
||||
vmem = psutil.virtual_memory()
|
||||
vmem_total = vmem.total / gb
|
||||
vmem_available = vmem.available / gb
|
||||
vmem_used = vmem_total - vmem_available
|
||||
smem = psutil.swap_memory()
|
||||
swap_total = smem.total / gb
|
||||
swap_free = smem.free / gb
|
||||
swap_used = smem.used / gb
|
||||
# https://psutil.readthedocs.io/en/latest/#psutil.cpu_percent
|
||||
cpu_util = psutil.cpu_percent(interval=1e-6)
|
||||
return CpuStats(
|
||||
total_virtual_mem=vmem_total,
|
||||
unallocated_virtual_mem=vmem_available,
|
||||
allocated_virtual_mem=vmem_used,
|
||||
total_swap_mem=swap_total,
|
||||
unallocated_swap_mem=swap_free,
|
||||
allocated_swap_mem=swap_used,
|
||||
utilization=cpu_util,
|
||||
)
|
||||
|
||||
|
||||
def cpu_stats_str(stats: CpuStats | None = None) -> str:
|
||||
"""Returns a string representation of the CPU stats."""
|
||||
|
||||
if stats is None:
|
||||
stats = cpu_stats()
|
||||
total, occupied, unused = (
|
||||
stats.total_virtual_mem,
|
||||
stats.allocated_virtual_mem,
|
||||
stats.unallocated_virtual_mem,
|
||||
)
|
||||
virtual_mem = (
|
||||
f"CPU virtual memory: {total:.2f}({occupied=:.2f}, {unused=:.2f}) GB"
|
||||
)
|
||||
total, occupied, unused = (
|
||||
stats.total_swap_mem,
|
||||
stats.allocated_swap_mem,
|
||||
stats.unallocated_swap_mem,
|
||||
)
|
||||
swap_mem = f"CPU swap memory: {total:.2f}({occupied=:.2f}, {unused=:.2f}) GB"
|
||||
percent = stats.utilization
|
||||
return f"{virtual_mem} {swap_mem} CPU Utilization: {percent:.2f}%"
|
||||
|
||||
|
||||
def init_partial_state(
|
||||
timeout: datetime.timedelta = datetime.timedelta(seconds=600),
|
||||
) -> None:
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
"""Fileutil lib to copy files between gcs and local."""
|
||||
|
||||
import filecmp
|
||||
import fnmatch
|
||||
import os
|
||||
import pathlib
|
||||
import shutil
|
||||
import subprocess
|
||||
import time
|
||||
from typing import List, Optional, Tuple
|
||||
import uuid
|
||||
|
||||
@@ -57,6 +60,96 @@ def force_gcs_path(uri: str) -> str:
|
||||
return uri
|
||||
|
||||
|
||||
def is_file_available(
|
||||
file_path: str, retry_interval_secs: int = 60, timeout_secs: int = 3600
|
||||
) -> bool:
|
||||
"""Checks and waits for a file to be available in GCS.
|
||||
|
||||
Args:
|
||||
file_path: The file path to check.
|
||||
retry_interval_secs: The interval in seconds to check the file.
|
||||
timeout_secs: The timeout in seconds to wait for the file.
|
||||
|
||||
Returns:
|
||||
True if the file is available, False otherwise.
|
||||
"""
|
||||
start_time = time.time()
|
||||
while True:
|
||||
try:
|
||||
file_check_cmd = ['gcloud', 'storage', 'ls', file_path]
|
||||
result = subprocess.run(
|
||||
file_check_cmd, capture_output=True, text=True, check=True
|
||||
)
|
||||
if file_path in result.stdout:
|
||||
logging.info('File %s exists.', file_path)
|
||||
return True
|
||||
except subprocess.CalledProcessError as e:
|
||||
elapsed_time = time.time() - start_time
|
||||
if elapsed_time > timeout_secs:
|
||||
logging.info(
|
||||
"Timeout: File '%s' not found after %d seconds. Error: %s",
|
||||
file_path,
|
||||
elapsed_time,
|
||||
e,
|
||||
)
|
||||
return False
|
||||
|
||||
logging.info(
|
||||
"File '%s' not found yet. Checking again in %d seconds. Error: %s",
|
||||
file_path,
|
||||
retry_interval_secs,
|
||||
e,
|
||||
)
|
||||
time.sleep(retry_interval_secs)
|
||||
|
||||
|
||||
def compare_dirs(
|
||||
local_dir: str,
|
||||
gcsfuse_dir: str,
|
||||
retry_interval_secs: int = 30,
|
||||
timeout_secs: int = 3600,
|
||||
) -> bool:
|
||||
"""Compares two directories and returns True if they are the same.
|
||||
|
||||
Args:
|
||||
local_dir: The local directory.
|
||||
gcsfuse_dir: The gcsfuse directory.
|
||||
retry_interval_secs: The interval in seconds to check the directories.
|
||||
timeout_secs: The timeout in seconds to wait for the directories.
|
||||
|
||||
Returns:
|
||||
True if the directories are the same, False otherwise.
|
||||
"""
|
||||
start_time = time.time()
|
||||
while True:
|
||||
if os.path.exists(local_dir) and os.path.exists(gcsfuse_dir):
|
||||
comparison = filecmp.dircmp(local_dir, gcsfuse_dir)
|
||||
if (
|
||||
not comparison.left_only
|
||||
and not comparison.right_only
|
||||
and not comparison.diff_files
|
||||
):
|
||||
return True
|
||||
elapsed_time = time.time() - start_time
|
||||
if elapsed_time > timeout_secs:
|
||||
logging.info(
|
||||
"Timeout: Directories '%s' and '%s' do not match after %d seconds.",
|
||||
local_dir,
|
||||
gcsfuse_dir,
|
||||
elapsed_time,
|
||||
)
|
||||
return False
|
||||
|
||||
logging.info(
|
||||
"Directories '%s' and '%s' do not match yet. Checking again in %d"
|
||||
' seconds.',
|
||||
local_dir,
|
||||
gcsfuse_dir,
|
||||
retry_interval_secs,
|
||||
)
|
||||
time.sleep(retry_interval_secs)
|
||||
|
||||
|
||||
def download_gcs_file_to_memory(gcs_uri: str) -> bytes:
|
||||
"""Downloads a gcs file to in memory.
|
||||
|
||||
@@ -352,3 +445,15 @@ def get_output_video_file(video_output_file_path: str) -> str:
|
||||
file_extension, '_overlay' + file_extension
|
||||
)
|
||||
return out_local_video_file_name
|
||||
|
||||
|
||||
def delete_local_file(local_file_path: str) -> None:
|
||||
"""Deletes a local file."""
|
||||
if os.path.exists(local_file_path):
|
||||
os.remove(local_file_path)
|
||||
|
||||
|
||||
def delete_local_dir(local_dir: str) -> None:
|
||||
"""Deletes a local directory recursively."""
|
||||
if os.path.exists(local_dir):
|
||||
shutil.rmtree(local_dir)
|
||||
|
||||
+119
@@ -0,0 +1,119 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# This launcher downloads model files from GCS to local model directory before
|
||||
# launching the actual command.
|
||||
#
|
||||
# If GCS URI is passed as an environment variable, set GCS_URI_ENV_KEY to the
|
||||
# environment variable name.
|
||||
# If GCS URI is passed as an argument, set GCS_URI_ARG_KEY to the argument name.
|
||||
# The argument must be in the format of '--$GCS_URI_ARG_KEY=gs://*'. Do not
|
||||
# separate argument name and value with spaces.
|
||||
# This script will also try reading from AIP_STORAGE_URI or AIP_STORAGE_DIR.
|
||||
# Note that AIP_STORAGE_DIR is expected to be a local path, so it bypasses the
|
||||
# download process.
|
||||
#
|
||||
# Input priority: AIP_STORAGE_DIR > AIP_STORAGE_URI > GCS_URI_ENV_KEY > GCS_URI_ARG_KEY.
|
||||
# Will output the local model directory to GCS_URI_ENV_KEY and GCS_URI_ARG_KEY
|
||||
# if they are set. Both will be updated if both set.
|
||||
#
|
||||
# Requires google-cloud-sdk as a dependency (for gcloud storage CLI).
|
||||
|
||||
set -e
|
||||
|
||||
readonly LOCAL_MODEL_DIR=${LOCAL_MODEL_DIR:-"/tmp/model_dir"}
|
||||
readonly LOCAL_ARGS_FILE=${LOCAL_ARGS_FILE:-"/tmp/args.txt"}
|
||||
|
||||
update_model_id() {
|
||||
if [[ ! -z "$GCS_URI_ENV_KEY" ]]; then
|
||||
echo "Updating env var $GCS_URI_ENV_KEY to $AIP_STORAGE_DIR."
|
||||
export "$GCS_URI_ENV_KEY"="$AIP_STORAGE_DIR"
|
||||
fi
|
||||
|
||||
if [[ ! -z "$GCS_URI_ARG_KEY" ]]; then
|
||||
echo "Updating args $GCS_URI_ARG_KEY to $AIP_STORAGE_DIR."
|
||||
updated=0
|
||||
for (( i=1; i <= $#; i++)); do
|
||||
arg="${!i}"
|
||||
if [[ "$arg" == "--$GCS_URI_ARG_KEY="* ]]; then
|
||||
echo "Found $arg, updating to $AIP_STORAGE_DIR."
|
||||
set -- "${@:1:(($i-1))}" "--$GCS_URI_ARG_KEY=$AIP_STORAGE_DIR" "${@:$(($i+1))}";
|
||||
updated=1
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [[ $updated -eq 0 ]]; then
|
||||
echo "Appending args $GCS_URI_ARG_KEY to $AIP_STORAGE_DIR."
|
||||
set -- "$@" "--$GCS_URI_ARG_KEY=$AIP_STORAGE_DIR";
|
||||
fi
|
||||
fi
|
||||
echo "$*" > "$LOCAL_ARGS_FILE"
|
||||
}
|
||||
|
||||
maybe_download_model() {
|
||||
if [[ -z "$GCS_URI_ENV_KEY" ]] && [[ -z "$GCS_URI_ARG_KEY" ]]; then
|
||||
echo "Internal error: Required GCS_URI_ENV_KEY or GCS_URI_ARG_KEY."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "$*" > "$LOCAL_ARGS_FILE"
|
||||
gcs_uri=""
|
||||
if [[ ! -z "$AIP_STORAGE_DIR" ]]; then
|
||||
# AIP_STORAGE_DIR is expected to be a local path.
|
||||
echo "AIP_STORAGE_DIR set, proceeding to run the launcher."
|
||||
update_model_id "$@"
|
||||
return
|
||||
elif [[ $AIP_STORAGE_URI == gs://* ]]; then
|
||||
# Check AIP_STORAGE_URI environment variable.
|
||||
echo "AIP_STORAGE_URI set and starts with 'gs://', proceeding to download from GCS."
|
||||
gcs_uri="$AIP_STORAGE_URI"
|
||||
elif [[ ! -z "$GCS_URI_ENV_KEY" ]] && [[ ${!GCS_URI_ENV_KEY} == gs://* ]]; then
|
||||
# Check custom environment variable.
|
||||
echo "Custom environment variable ${GCS_URI_ENV_KEY} set and starts with 'gs://', proceeding to download from GCS."
|
||||
gcs_uri="${!GCS_URI_ENV_KEY}"
|
||||
elif [[ ! -z "$GCS_URI_ARG_KEY" ]]; then
|
||||
# Check custom args.
|
||||
for arg in "$@"; do
|
||||
if [[ "$arg" == "--$GCS_URI_ARG_KEY=gs://"* ]]; then
|
||||
gcs_uri="${arg#*=}"
|
||||
echo "Custom args ${GCS_URI_ARG_KEY} set and starts with 'gs://', proceeding to download from GCS."
|
||||
break
|
||||
elif [[ "$arg" == "--$GCS_URI_ARG_KEY" ]]; then
|
||||
echo "Found $GCS_URI_ARG_KEY, but it's not in the format of '--$GCS_URI_ARG_KEY=gs://*'."
|
||||
echo "Ensure the value of $GCS_URI_ARG_KEY is within the same arg, separated by '='."
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
if [[ -z "$gcs_uri" ]]; then
|
||||
echo "No GCS URI found, proceeding to run the launcher."
|
||||
return
|
||||
fi
|
||||
|
||||
# Remove trailing '/' if any.
|
||||
gcs_uri="${gcs_uri%%/}"
|
||||
export AIP_STORAGE_DIR="$LOCAL_MODEL_DIR/${gcs_uri##gs://}"
|
||||
|
||||
# Create the target directory.
|
||||
mkdir -p "$AIP_STORAGE_DIR"
|
||||
echo "Downloading model from ${gcs_uri} to ${AIP_STORAGE_DIR}."
|
||||
|
||||
# Use gcloud storage CLI to copy the content from GCS to the target directory.
|
||||
if gcloud storage cp -r "$gcs_uri/*" "$AIP_STORAGE_DIR"; then
|
||||
echo "Model downloaded successfully to ${AIP_STORAGE_DIR}."
|
||||
update_model_id "$@"
|
||||
else
|
||||
echo "Failed to download model from GCS."
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
run_local_command() {
|
||||
command=$(cat "$LOCAL_ARGS_FILE")
|
||||
rm -f "$LOCAL_ARGS_FILE"
|
||||
echo "Launch command: $command"
|
||||
eval "$command"
|
||||
}
|
||||
|
||||
maybe_download_model "$@"
|
||||
run_local_command
|
||||
+92
-2
@@ -1,17 +1,107 @@
|
||||
"""Sync local directory to GCS directory using rsync."""
|
||||
|
||||
from collections.abc import Sequence
|
||||
import multiprocessing
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
from typing import Optional, Sequence, Tuple
|
||||
|
||||
from absl import logging
|
||||
|
||||
from util import constants
|
||||
from util import fileutils
|
||||
|
||||
_GCS_COMMAND_RETRIES = 3
|
||||
_RSYNC_RETRY_INTERVAL_SECS = 30
|
||||
|
||||
|
||||
def is_gcs_or_gcsfuse_path(path: str) -> bool:
|
||||
"""Returns if the path is a GCS or gcsfuse path.
|
||||
|
||||
Args:
|
||||
path: The path to check.
|
||||
|
||||
Returns:
|
||||
True if the path is a GCS or gcsfuse path.
|
||||
"""
|
||||
return path.startswith(
|
||||
(constants.GCS_URI_PREFIX, constants.GCSFUSE_URI_PREFIX)
|
||||
)
|
||||
|
||||
|
||||
def manage_sync_path(
|
||||
path: str, node_rank: Optional[int] = None
|
||||
) -> Tuple[str, str]:
|
||||
"""Returns local dir and GCS location for the given path if the given path is a GCS or gcsfuse path.
|
||||
|
||||
It will also create a local directory if it does not exist. Otherwise, it
|
||||
returns the same path.
|
||||
|
||||
Args:
|
||||
path: The local or GCS path to manage.
|
||||
node_rank: The node rank to be appended to the GCS path.
|
||||
|
||||
Returns:
|
||||
The local and GCS paths.
|
||||
"""
|
||||
local_dir = path
|
||||
gcs_dir = path
|
||||
if is_gcs_or_gcsfuse_path(path):
|
||||
local_dir = os.path.join(
|
||||
constants.LOCAL_OUTPUT_DIR,
|
||||
fileutils.force_gcs_fuse_path(path)[1:],
|
||||
)
|
||||
gcs_dir = fileutils.force_gcs_path(path)
|
||||
if not os.path.exists(local_dir):
|
||||
os.makedirs(local_dir, exist_ok=True)
|
||||
|
||||
if node_rank is None:
|
||||
return local_dir, gcs_dir
|
||||
return local_dir, os.path.join(gcs_dir, f"node-{node_rank}")
|
||||
|
||||
|
||||
def setup_gcs_rsync(
|
||||
dirs_to_sync: Sequence[Tuple[str, str]],
|
||||
mp_queue: multiprocessing.Queue,
|
||||
gcs_rsync_interval_secs: int,
|
||||
) -> multiprocessing.Process:
|
||||
"""Sets up the GCS rsync process.
|
||||
|
||||
Args:
|
||||
dirs_to_sync: The absolute directory paths which will be synced to GCS.
|
||||
mp_queue: The multiprocessing queue to check if the training is finished.
|
||||
gcs_rsync_interval_secs: Integer, interval in seconds to run gcs rsync.
|
||||
|
||||
Returns:
|
||||
The GCS rsync process.
|
||||
"""
|
||||
rsync_process = multiprocessing.Process(
|
||||
target=start_gcs_rsync,
|
||||
args=(dirs_to_sync, mp_queue, gcs_rsync_interval_secs),
|
||||
)
|
||||
rsync_process.start()
|
||||
return rsync_process
|
||||
|
||||
|
||||
def cleanup_gcs_rsync(
|
||||
rsync_process: multiprocessing.Process, mp_queue: multiprocessing.Queue
|
||||
) -> None:
|
||||
"""Cleans up the GCS rsync process.
|
||||
|
||||
Args:
|
||||
rsync_process: The GCS rsync process.
|
||||
mp_queue: The multiprocessing queue.
|
||||
"""
|
||||
mp_queue.put("finish rsync process")
|
||||
rsync_process.join()
|
||||
if rsync_process.exitcode == 0:
|
||||
logging.info("Artifacts have been uploaded to GCS.")
|
||||
else:
|
||||
logging.error(
|
||||
"GCS rsync process failed with exit code %d.", rsync_process.exitcode
|
||||
)
|
||||
|
||||
|
||||
def _rsync_local_to_gcs(local_dir: str, gcs_dir: str) -> None:
|
||||
"""Syncs the local directory to GCS.
|
||||
|
||||
@@ -57,7 +147,7 @@ def _rsync_local_to_gcs(local_dir: str, gcs_dir: str) -> None:
|
||||
|
||||
|
||||
def start_gcs_rsync(
|
||||
dirs_to_sync: Sequence[tuple[str, str]],
|
||||
dirs_to_sync: Sequence[Tuple[str, str]],
|
||||
mp_queue: multiprocessing.Queue,
|
||||
gcs_rsync_interval_secs: int,
|
||||
) -> None:
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
#!/bin/bash
|
||||
|
||||
# !/bin/bash
|
||||
# The Startup prober built to check whether models listed in local disk are
|
||||
# loaded in memory and are ready to serve traffic. The script returns 0 if
|
||||
# succeed. Any other returned value are consider as an error. More detail could be
|
||||
# found from [shell script Exit codes](http://shellscript.sh/exitcodes.html).
|
||||
#
|
||||
# TorchServe: The Management API listens on port 8081 and is only accessible
|
||||
# from localhost by default.
|
||||
|
||||
if [[ -z "${MNG_PORT}" ]]; then
|
||||
MNG_PORT=7081 # We default the management_port to 7081.
|
||||
else
|
||||
MNG_PORT="${MNG_PORT}"
|
||||
fi
|
||||
|
||||
check_model_availability(){
|
||||
local MODEL_NAME=$1
|
||||
# Returns whether "READY" is found in the model status.
|
||||
# Reference: https://pytorch.org/serve/management_api.html#describe-model.
|
||||
curl -s "http://localhost:${MNG_PORT}/models/${MODEL_NAME}" | grep "READY" -q
|
||||
}
|
||||
|
||||
main(){
|
||||
check_model_availability "$MODEL" # Assume Dockerfile sets MODEL environment parameter.
|
||||
local available=$?
|
||||
if [[ $available -gt 0 ]]
|
||||
then
|
||||
echo "Warning: Model(${MODEL}) is not yet available."
|
||||
return 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
main
|
||||
@@ -26,6 +26,8 @@
|
||||
/vertex_endpoints/find_ideal_machine_type/find_ideal_machine_type/find_ideal_machine_type.ipynb @entrpn
|
||||
/vertex_endpoints/nvidia-triton/nvidia-triton-custom-container-prediction.ipynb @RajeshThallam
|
||||
/vertex_endpoints/optimized_tensorflow_runtime @vlasenkoalexey
|
||||
/notebooks/community/alphagenome/cloudai_alphagenome_vai_quickstart.ipynb @dpanigra
|
||||
/notebooks/community/weathernext/weathernext_2_early_access_program.ipynb @dpanigra
|
||||
/notebooks/community/ml_ops/stage2/get_started_with_visionapi_and_automl.ipynb @mansari
|
||||
/notebooks/community/neo4j/graph_paysim.ipynb @benofben @laeg
|
||||
/notebooks/community/ml_ops/stage1/get_started_with_visionapi_and_vertex_datasets.ipynb @mansari
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||

|
||||
|
||||
# AlphaGenome
|
||||
[**Overview**](#overview) | [**Use Cases**](#use-cases) | [**Documentation**](#documentation) | [**Pricing**](#pricing) | [**Quick start**](#quick-start)
|
||||
|
||||
## Overview
|
||||
**Disclaimer:** *Experimental*.
|
||||
|
||||
*The AlphaGenome Private Preview is a "Pre-GA Offering" subject to the "Pre-GA
|
||||
Offerings Terms" in the General Service Terms section of the Google Cloud
|
||||
[Service Specific Terms](https://cloud.google.com/terms/service-terms). It is
|
||||
also a “Generative AI Preview Product” as defined in and subject to the
|
||||
[Additional Terms for Generative AI Preview Products](https://cloud.google.com/trustedtester/aitos?e=48754805&hl=en).
|
||||
Pre-GA products are available "as is" and might have limited support. For more
|
||||
information, see the [launch stage](https://cloud.google.com/products?e=48754805#product-launch-stages)
|
||||
descriptions.*
|
||||
|
||||
Access to the AlphaGenome model capabilities requires application and approval.
|
||||
Users must be added to an allowlist to use the service.
|
||||
If you are interested in applying to the program, **Request Access** above.
|
||||
|
||||
|
||||
|
||||
AlphaGenome is Google DeepMind’s unifying model for deciphering the regulatory
|
||||
code within DNA sequences.
|
||||
|
||||
AlphaGenome offers multimodal predictions, encompassing diverse functional
|
||||
outputs such as gene expression, splicing patterns, chromatin features, and
|
||||
contact maps (see diagram below). The model analyzes DNA sequences of up to 1
|
||||
million base pairs in length and can deliver predictions at single base-pair
|
||||
resolution for most outputs.
|
||||
|
||||
Training data was sourced from large public consortia including
|
||||
[ENCODE](http://encodeproject.org/), [GTEx](https://www.gtexportal.org/),
|
||||
[4D Nucleome](https://4dnucleome.org/) and
|
||||
[FANTOM5](https://fantom.gsc.riken.jp/5/), which experimentally measured these
|
||||
properties covering important modalities of gene regulation across hundreds of
|
||||
human and mouse cell types and tissues.
|
||||
|
||||

|
||||
|
||||
## Use Cases
|
||||
* **Sequence-to-function predictions:** Predict multiple functional tracks (such as gene expression, splicing) from DNA sequences across a wide variety of tissues and cell types.
|
||||
|
||||
* **Variant effect scoring:** Assess the impact of genetic variants by comparing predictions for the reference and alternative alleles and summarising the differences between them.
|
||||
|
||||
* **Identify functional regions:** Use in silico mutagenesis (ISM) to identify functionally important regions in the DNA sequence.
|
||||
|
||||
* **Human and mouse capability:** Generate predictions for both human and mouse genomes.
|
||||
|
||||
## Documentation
|
||||
This API provides access to AlphaGenome, Google DeepMind's unifying model for
|
||||
deciphering the regulatory code within DNA sequences. AlphaGenome offers
|
||||
multimodal predictions, encompassing diverse functional outputs including gene
|
||||
expression, splicing patterns, chromatin features, and contact maps (see diagram
|
||||
below). The model analyzes up to 1 million base pairs of DNA sequence and can
|
||||
deliver predictions at single base-pair resolution for most modalities.
|
||||
AlphaGenome achieves state-of-the-art performance across a range of genomic
|
||||
prediction benchmarks, including diverse variant effect prediction tasks.
|
||||
|
||||
The Google Cloud API for AlphaGenome provides a way for Google Cloud customers
|
||||
to explore the AlphaGenome API for commercial use cases. This API is in private
|
||||
preview (Request Access above). Once allowlisted, customers can access the API
|
||||
directly or use the [colab](cloudai_alphagenome_vai_quickstart.ipynb).
|
||||
|
||||
### Acknowledgements
|
||||
|
||||
*Avsec, Ž., Latysheva, N., Cheng, J., Novati, G., Taylor, K. R., Ward, T., ... Kohli, P. (2025). AlphaGenome: advancing regulatory variant effect prediction with a unified DNA sequence model. bioRxiv.* [https://doi.org/10.1101/2025.06.25.661532](https://doi.org/10.1101/2025.06.25.661532)
|
||||
|
||||
### Contact
|
||||
If you have any questions on using these models on Google Cloud please contact:
|
||||
[alphagenome-cloud-external@google.com](mailto:alphagenome-cloud-external@google.com) or join the community [Discourse](https://www.alphagenomecommunity.com/) for more generic questions on AlphaGenome.
|
||||
|
||||
### Links
|
||||
|
||||
* Read our [paper](https://doi.org/10.1101/2025.06.25.661532)
|
||||
* Read our [blog post](https://deepmind.google/discover/blog/alphagenome-ai-for-better-understanding-the-genome)
|
||||
* Join the [community](https://www.alphagenomecommunity.com/)
|
||||
* Check out the [AlphaGenome 101 Video](https://youtu.be/Xbvloe13nak)
|
||||
|
||||
## Pricing
|
||||
Access to AlphaGenome on Vertex AI is currently restricted.
|
||||
To utilize these models via this service:
|
||||
|
||||
* You must **Request Access** using your Google contact.
|
||||
* Your application will be reviewed, and if approved, you will be **added to
|
||||
an allowlist**.
|
||||
* Only allowlisted users can access the API
|
||||
* **Pricing information** will be shared directly with users upon approval
|
||||
and placement on the allowlist.
|
||||
|
||||
## Quick start
|
||||
The quickest way to get started with the AlphaGenome in Google Cloud Platform is to run [our example notebook](cloudai_alphagenome_vai_quickstart.ipynb) in [Google Colab](https://colab.research.google.com/).
|
||||
File diff suppressed because one or more lines are too long
+3
-3
@@ -37,18 +37,18 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/bigquery_ml/Anomaly_detection_in_Cloud_Audit_logs_with_BQML.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/bigquery_ml/Anomaly_detection_in_Cloud_Audit_logs_with_BQML.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/bigquery_ml/Anomaly_detection_in_Cloud_Audit_logs_with_BQML.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
|
||||
@@ -47,18 +47,18 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/bigquery_ml/bq_ml_with_vision_translation_nlp.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/bigquery_ml/bq_ml_with_vision_translation_nlp.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/bigquery_ml/bq_ml_with_vision_translation_nlp.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -35,18 +35,18 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/bigquery_ml/bqml-online-prediction.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/bigquery_ml/bqml-online-prediction.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/official/bigquery_ml/bqml-online-prediction.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
@@ -359,7 +359,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l {REGION} -p {PROJECT_ID} {BUCKET_URI}"
|
||||
"! gcloud storage buckets create --location={REGION} --project={PROJECT_ID} {BUCKET_URI}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1098,7 +1098,7 @@
|
||||
" ! bq rm -r -f $PROJECT_ID:$BQ_DATASET_NAME\n",
|
||||
"# delete the Cloud Storage bucket\n",
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil -m rm -r $BUCKET_URI"
|
||||
" ! gcloud storage rm --recursive $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -38,7 +38,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/cohere/cohere_embedding_with_matching_engine.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -449,7 +449,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gcloud storage buckets create --location=$REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -470,7 +470,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# this will not return anything if the bucket is empty\n",
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -588,7 +588,7 @@
|
||||
"# NOTE: Everything in this GCS DIR will be DELETED before uploading the data.\n",
|
||||
"# A CommandException is expected if no data is present\n",
|
||||
"\n",
|
||||
"! gsutil rm -rf {BUCKET_NAME}/*"
|
||||
"! gcloud storage rm --recursive --continue-on-error {BUCKET_NAME}/*"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -599,7 +599,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil cp cohere_embeddings.json {BUCKET_NAME}/cohere_embeddings.json"
|
||||
"! gcloud storage cp cohere_embeddings.json {BUCKET_NAME}/cohere_embeddings.json"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -610,7 +610,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls {BUCKET_NAME}"
|
||||
"! gcloud storage ls {BUCKET_NAME}"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -36,18 +36,18 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/vertex_ai_experiments_classification.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/vertex_ai_experiments_classification.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/vertex_ai_experiments_classification.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
@@ -537,7 +537,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
|
||||
"! gcloud storage buckets create --location=$REGION --project=$PROJECT_ID $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -557,7 +557,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -627,9 +627,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
|
||||
"! gcloud storage buckets add-iam-policy-binding $BUCKET_URI --member=serviceAccount:{SERVICE_ACCOUNT} --role=roles/storage.objectCreator\n",
|
||||
"\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
|
||||
"! gcloud storage buckets add-iam-policy-binding $BUCKET_URI --member=serviceAccount:{SERVICE_ACCOUNT} --role=roles/storage.objectViewer"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1736,7 +1736,7 @@
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -rf {BUCKET_URI}"
|
||||
" ! gcloud storage rm --recursive --continue-on-error {BUCKET_URI}"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -33,18 +33,18 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/explainable_ai/SDK_Custom_Container_XAI.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/explainable_ai/SDK_Custom_Container_XAI.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/explainable_ai/SDK_Custom_Container_XAI.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
@@ -405,7 +405,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
|
||||
"! gcloud storage buckets create --location=$REGION --project=$PROJECT_ID $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -425,7 +425,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -698,7 +698,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!gsutil cp app/model.joblib {BUCKET_URI}/{MODEL_ARTIFACT_DIR}/"
|
||||
"!gcloud storage cp app/model.joblib {BUCKET_URI}/{MODEL_ARTIFACT_DIR}/"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1516,7 +1516,7 @@
|
||||
"\n",
|
||||
"delete_bucket = False\n",
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -rf {BUCKET_URI}"
|
||||
" ! gcloud storage rm --recursive --continue-on-error {BUCKET_URI}"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -637,20 +637,20 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Upload CSV data to Cloud Storage by passing gsutil commands to system\n",
|
||||
"# Upload CSV data to Cloud Storage by passing gcloud storage commands to system\n",
|
||||
"gcs_url <- paste0(\"gs://\", BUCKET_NAME, \"/\")\n",
|
||||
"\n",
|
||||
"command <- paste(\"gsutil mb\", gcs_url)\n",
|
||||
"command <- paste(\"gcloud storage buckets create\", gcs_url)\n",
|
||||
"\n",
|
||||
"system(command)\n",
|
||||
"\n",
|
||||
"gcs_data_dir <- paste0(\"gs://\", BUCKET_NAME, \"/data\")\n",
|
||||
"\n",
|
||||
"command <- paste(\"gsutil cp data/*_data.csv\", gcs_data_dir)\n",
|
||||
"command <- paste(\"gcloud storage cp data/*_data.csv\", gcs_data_dir)\n",
|
||||
"\n",
|
||||
"system(command)\n",
|
||||
"\n",
|
||||
"command <- paste(\"gsutil ls -l\", gcs_data_dir)\n",
|
||||
"command <- paste(\"gcloud storage ls --long\", gcs_data_dir)\n",
|
||||
"\n",
|
||||
"system(command, intern = TRUE)"
|
||||
]
|
||||
@@ -679,4 +679,4 @@
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 4
|
||||
}
|
||||
}
|
||||
@@ -33,12 +33,12 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/official/feature_store/gapic-feature-store.ipynb\"\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/official/feature_store/gapic-feature-store.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -34,20 +34,20 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage6/get_started_vertex_feature_store_serving.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" \n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage6/get_started_vertex_feature_store_serving.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" \n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage6/get_started_vertex_feature_store_serving.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -471,8 +471,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
"! gcloud storage buckets create --location=$REGION $BUCKET_URI" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -491,8 +490,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_URI" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -511,8 +509,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil uniformbucketlevelaccess set on {BUCKET_URI}"
|
||||
]
|
||||
"! gcloud storage buckets update --uniform-bucket-level-access {BUCKET_URI}" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1910,9 +1907,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil -m rm -r $BUCKET_URI\n",
|
||||
"! gsutil rb $BUCKET_URI"
|
||||
]
|
||||
"! gcloud storage rm --recursive $BUCKET_URI\n", "! gcloud storage buckets delete $BUCKET_URI" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
@@ -33,18 +33,18 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/vertex-ai-samples/blob/main/notebooks/community/feature_store/mobile_gaming/mobile_gaming_feature_store.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/feature_store/mobile_gaming/mobile_gaming_feature_store.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/feature_store/mobile_gaming/mobile_gaming_feature_store.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" <img src=\"https://www.gstatic.com/images/branding/gcpiconscolors/vertexai/v1/32px.svg\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
|
||||
@@ -32,12 +32,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/feature_store/sdk-feature-store.ipynb\"\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/feature_store/sdk-feature-store.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -421,7 +421,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gcloud storage buckets create --location=$REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -441,7 +441,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -478,9 +478,7 @@
|
||||
"import time\n",
|
||||
"\n",
|
||||
"from google.cloud.aiplatform import gapic as aip\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.json_format import MessageToJson, ParseDict\n",
|
||||
"from google.protobuf.struct_pb2 import Struct, Value"
|
||||
"from google.protobuf import json_format"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -886,11 +884,11 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
"! gcloud storage cat $FILE | head"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1329,7 +1327,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n2\n",
|
||||
"test_items = !gcloud storage cat $IMPORT_FILE | head -n2\n",
|
||||
"if len(str(test_items[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
@@ -1363,8 +1361,8 @@
|
||||
"file_1 = test_item_1.split(\"/\")[-1]\n",
|
||||
"file_2 = test_item_2.split(\"/\")[-1]\n",
|
||||
"\n",
|
||||
"! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"! gcloud storage cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gcloud storage cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"\n",
|
||||
"test_item_1 = BUCKET_NAME + \"/\" + file_1\n",
|
||||
"test_item_2 = BUCKET_NAME + \"/\" + file_2"
|
||||
@@ -1408,7 +1406,7 @@
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
"! gcloud storage cat $gcs_input_uri"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1692,8 +1690,8 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def get_latest_predictions(gcs_out_dir):\n",
|
||||
" \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gsutil ls $gcs_out_dir\n",
|
||||
" \"\"\"Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gcloud storage ls $gcs_out_dir\n",
|
||||
" latest = \"\"\n",
|
||||
" for folder in folders:\n",
|
||||
" subfolder = folder.split(\"/\")[-2]\n",
|
||||
@@ -1711,9 +1709,9 @@
|
||||
" raise Exception(\"Batch Job Failed\")\n",
|
||||
" else:\n",
|
||||
" folder = get_latest_predictions(predictions)\n",
|
||||
" ! gsutil ls $folder/prediction*.jsonl\n",
|
||||
" ! gcloud storage ls $folder/prediction*.jsonl\n",
|
||||
"\n",
|
||||
" ! gsutil cat $folder/prediction*.jsonl\n",
|
||||
" ! gcloud storage cat $folder/prediction*.jsonl\n",
|
||||
" break\n",
|
||||
" time.sleep(60)"
|
||||
]
|
||||
@@ -1808,7 +1806,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+10
-12
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_export_edge.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_export_edge.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -416,7 +416,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gcloud storage buckets create --location=$REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -436,7 +436,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -474,7 +474,6 @@
|
||||
"\n",
|
||||
"from google.cloud.aiplatform import gapic as aip\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.json_format import MessageToJson, ParseDict\n",
|
||||
"from google.protobuf.struct_pb2 import Struct, Value"
|
||||
]
|
||||
},
|
||||
@@ -784,11 +783,11 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
"! gcloud storage cat $FILE | head"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1215,7 +1214,6 @@
|
||||
" }\n",
|
||||
" response = clients[\"model\"].export_model(name=name, output_config=output_config)\n",
|
||||
" print(\"Long running operation:\", response.operation.name)\n",
|
||||
" result = response.result(timeout=1800)\n",
|
||||
" metadata = response.operation.metadata\n",
|
||||
" artifact_uri = str(metadata.value).split(\"\\\\\")[-1][4:-1]\n",
|
||||
" print(\"Artifact Uri\", artifact_uri)\n",
|
||||
@@ -1244,9 +1242,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $model_package\n",
|
||||
"! gcloud storage ls $model_package\n",
|
||||
"# Download the model artifacts\n",
|
||||
"! gsutil cp -r $model_package tflite\n",
|
||||
"! gcloud storage cp --recursive $model_package tflite\n",
|
||||
"\n",
|
||||
"tflite_path = \"tflite/model.tflite\""
|
||||
]
|
||||
@@ -1305,7 +1303,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"test_items = ! gcloud storage cat $IMPORT_FILE | head -n1\n",
|
||||
"test_item = test_items[0].split(\",\")[0]\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
@@ -1449,7 +1447,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -721,12 +721,10 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
]
|
||||
"! gcloud storage cat $FILE | head" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1342,8 +1340,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_item = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"if len(str(test_item[0]).split(\",\")) == 3:\n",
|
||||
"test_item = !gcloud storage cat $IMPORT_FILE | head -n1\n", "if len(str(test_item[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
"else:\n",
|
||||
" test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
@@ -1568,8 +1565,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+6
-10
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_online_proxy.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_classification_online_proxy.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -721,12 +721,10 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
]
|
||||
"! gcloud storage cat $FILE | head" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1342,8 +1340,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_item = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"if len(str(test_item[0]).split(\",\")) == 3:\n",
|
||||
"test_item = !gcloud storage cat $IMPORT_FILE | head -n1\n", "if len(str(test_item[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
"else:\n",
|
||||
" test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
@@ -1747,8 +1744,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+16
-18
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_object_detection_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_object_detection_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -421,7 +421,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gcloud storage buckets create --location=$REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -441,7 +441,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -478,9 +478,7 @@
|
||||
"import time\n",
|
||||
"\n",
|
||||
"from google.cloud.aiplatform import gapic as aip\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.json_format import MessageToJson, ParseDict\n",
|
||||
"from google.protobuf.struct_pb2 import Struct, Value"
|
||||
"from google.protobuf import json_format"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -887,11 +885,11 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
"! gcloud storage cat $FILE | head"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1333,7 +1331,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n2\n",
|
||||
"test_items = !gcloud storage cat $IMPORT_FILE | head -n2\n",
|
||||
"cols_1 = str(test_items[0]).split(\",\")\n",
|
||||
"cols_2 = str(test_items[1]).split(\",\")\n",
|
||||
"if len(cols_1) == 11:\n",
|
||||
@@ -1373,8 +1371,8 @@
|
||||
"file_1 = test_item_1.split(\"/\")[-1]\n",
|
||||
"file_2 = test_item_2.split(\"/\")[-1]\n",
|
||||
"\n",
|
||||
"! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"! gcloud storage cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gcloud storage cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"\n",
|
||||
"test_item_1 = BUCKET_NAME + \"/\" + file_1\n",
|
||||
"test_item_2 = BUCKET_NAME + \"/\" + file_2"
|
||||
@@ -1418,7 +1416,7 @@
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
"! gcloud storage cat $gcs_input_uri"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1704,8 +1702,8 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"def get_latest_predictions(gcs_out_dir):\n",
|
||||
" \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gsutil ls $gcs_out_dir\n",
|
||||
" \"\"\"Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gcloud storage ls $gcs_out_dir\n",
|
||||
" latest = \"\"\n",
|
||||
" for folder in folders:\n",
|
||||
" subfolder = folder.split(\"/\")[-2]\n",
|
||||
@@ -1723,9 +1721,9 @@
|
||||
" raise Exception(\"Batch Job Failed\")\n",
|
||||
" else:\n",
|
||||
" folder = get_latest_predictions(predictions)\n",
|
||||
" ! gsutil ls $folder/prediction*.jsonl\n",
|
||||
" ! gcloud storage ls $folder/prediction*.jsonl\n",
|
||||
"\n",
|
||||
" ! gsutil cat $folder/prediction*.jsonl\n",
|
||||
" ! gcloud storage cat $folder/prediction*.jsonl\n",
|
||||
" break\n",
|
||||
" time.sleep(60)"
|
||||
]
|
||||
@@ -1820,7 +1818,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+11
-14
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_object_detection_export_edge.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_object_detection_export_edge.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -416,7 +416,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -436,7 +436,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -473,9 +473,7 @@
|
||||
"import time\n",
|
||||
"\n",
|
||||
"from google.cloud.aiplatform import gapic as aip\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.json_format import MessageToJson, ParseDict\n",
|
||||
"from google.protobuf.struct_pb2 import Struct, Value"
|
||||
"from google.protobuf import json_format\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -785,11 +783,11 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
"! gcloud storage cat $FILE | head"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1218,7 +1216,6 @@
|
||||
" }\n",
|
||||
" response = clients[\"model\"].export_model(name=name, output_config=output_config)\n",
|
||||
" print(\"Long running operation:\", response.operation.name)\n",
|
||||
" result = response.result(timeout=1800)\n",
|
||||
" metadata = response.operation.metadata\n",
|
||||
" artifact_uri = str(metadata.value).split(\"\\\\\")[-1][4:-1]\n",
|
||||
" print(\"Artifact Uri\", artifact_uri)\n",
|
||||
@@ -1247,9 +1244,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls $model_package\n",
|
||||
"! gcloud storage ls $model_package\n",
|
||||
"# Download the model artifacts\n",
|
||||
"! gsutil cp -r $model_package tflite\n",
|
||||
"! gcloud storage cp --recursive $model_package tflite\n",
|
||||
"\n",
|
||||
"tflite_path = \"tflite/model.tflite\""
|
||||
]
|
||||
@@ -1308,7 +1305,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"test_items = ! gcloud storage cat $IMPORT_FILE | head -n1\n",
|
||||
"test_item = test_items[0].split(\",\")[0]\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
@@ -1452,7 +1449,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+7
-9
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_object_detection_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_object_detection_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -396,9 +396,7 @@
|
||||
"import time\n",
|
||||
"\n",
|
||||
"from google.cloud.aiplatform import gapic as aip\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.json_format import MessageToJson, ParseDict\n",
|
||||
"from google.protobuf.struct_pb2 import Struct, Value"
|
||||
"from google.protobuf import json_format"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -722,11 +720,11 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
"! gcloud storage cat $FILE | head"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1345,7 +1343,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"test_items = !gcloud storage cat $IMPORT_FILE | head -n1\n",
|
||||
"cols = str(test_items[0]).split(\",\")\n",
|
||||
"if len(cols) == 11:\n",
|
||||
" test_item = str(cols[1])\n",
|
||||
@@ -1574,7 +1572,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_segmentation_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_segmentation_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -421,8 +421,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -441,8 +440,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -891,12 +889,10 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
]
|
||||
"! gcloud storage cat $FILE | head" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1325,8 +1321,7 @@
|
||||
"source": [
|
||||
"import json\n",
|
||||
"\n",
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n2\n",
|
||||
"test_data_1 = test_items[0].replace(\"'\", '\"')\n",
|
||||
"test_items = !gcloud storage cat $IMPORT_FILE | head -n2\n", "test_data_1 = test_items[0].replace(\"'\", '\"')\n",
|
||||
"test_data_1 = json.loads(test_data_1)\n",
|
||||
"test_data_2 = test_items[0].replace(\"'\", '\"')\n",
|
||||
"test_data_2 = json.loads(test_data_2)\n",
|
||||
@@ -1367,9 +1362,8 @@
|
||||
"file_1 = test_item_1.split(\"/\")[-1]\n",
|
||||
"file_2 = test_item_2.split(\"/\")[-1]\n",
|
||||
"\n",
|
||||
"! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"\n",
|
||||
"! gcloud storage cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gcloud storage cp $test_item_2 $BUCKET_NAME/$file_2\n", "\n",
|
||||
"test_item_1 = BUCKET_NAME + \"/\" + file_1\n",
|
||||
"test_item_2 = BUCKET_NAME + \"/\" + file_2"
|
||||
]
|
||||
@@ -1412,8 +1406,7 @@
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
"! gcloud storage cat $gcs_input_uri" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1681,8 +1674,7 @@
|
||||
"source": [
|
||||
"def get_latest_predictions(gcs_out_dir):\n",
|
||||
" \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gsutil ls $gcs_out_dir\n",
|
||||
" latest = \"\"\n",
|
||||
" folders = !gcloud storage ls $gcs_out_dir\n", " latest = \"\"\n",
|
||||
" for folder in folders:\n",
|
||||
" subfolder = folder.split(\"/\")[-2]\n",
|
||||
" if subfolder.startswith(\"prediction-\"):\n",
|
||||
@@ -1699,10 +1691,8 @@
|
||||
" raise Exception(\"Batch Job Failed\")\n",
|
||||
" else:\n",
|
||||
" folder = get_latest_predictions(predictions)\n",
|
||||
" ! gsutil ls $folder/prediction*.jsonl\n",
|
||||
"\n",
|
||||
" ! gsutil cat $folder/prediction*.jsonl\n",
|
||||
" break\n",
|
||||
" ! gcloud storage ls $folder/prediction*.jsonl\n", "\n",
|
||||
" ! gcloud storage cat $folder/prediction*.jsonl\n", " break\n",
|
||||
" time.sleep(60)"
|
||||
]
|
||||
},
|
||||
@@ -1796,8 +1786,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_segmentation_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_image_segmentation_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -726,12 +726,10 @@
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $FILE | head"
|
||||
]
|
||||
"! gcloud storage cat $FILE | head" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1338,8 +1336,7 @@
|
||||
"source": [
|
||||
"import json\n",
|
||||
"\n",
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"test_data = test_items[0].replace(\"'\", '\"')\n",
|
||||
"test_items = !gcloud storage cat $IMPORT_FILE | head -n1\n", "test_data = test_items[0].replace(\"'\", '\"')\n",
|
||||
"test_data = json.loads(test_data)\n",
|
||||
"try:\n",
|
||||
" test_item = test_data[\"image_gcs_uri\"]\n",
|
||||
@@ -1555,8 +1552,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+12
-22
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -421,8 +421,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -441,8 +440,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -809,14 +807,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n", "\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n", "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
" raise Exception(\"label column missing\")"
|
||||
@@ -1377,8 +1372,7 @@
|
||||
" f.write(str(INSTANCE_2) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
"! gcloud storage cat $gcs_input_uri" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1646,8 +1640,7 @@
|
||||
"source": [
|
||||
"def get_latest_predictions(gcs_out_dir):\n",
|
||||
" \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gsutil ls $gcs_out_dir\n",
|
||||
" latest = \"\"\n",
|
||||
" folders = !gcloud storage ls $gcs_out_dir\n", " latest = \"\"\n",
|
||||
" for folder in folders:\n",
|
||||
" subfolder = folder.split(\"/\")[-2]\n",
|
||||
" if subfolder.startswith(\"prediction-\"):\n",
|
||||
@@ -1664,10 +1657,8 @@
|
||||
" raise Exception(\"Batch Job Failed\")\n",
|
||||
" else:\n",
|
||||
" folder = get_latest_predictions(predictions)\n",
|
||||
" ! gsutil ls $folder/prediction*.csv\n",
|
||||
"\n",
|
||||
" ! gsutil cat $folder/prediction*.csv\n",
|
||||
" break\n",
|
||||
" ! gcloud storage ls $folder/prediction*.csv\n", "\n",
|
||||
" ! gcloud storage cat $folder/prediction*.csv\n", " break\n",
|
||||
" time.sleep(60)"
|
||||
]
|
||||
},
|
||||
@@ -1761,8 +1752,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+6
-10
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -735,14 +735,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n", "\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n", "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
" raise Exception(\"label column missing\")"
|
||||
@@ -1668,8 +1665,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+12
-22
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -421,8 +421,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -441,8 +440,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -810,14 +808,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n", "\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n", "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
" raise Exception(\"label column missing\")"
|
||||
@@ -1363,8 +1358,7 @@
|
||||
" f.write(str(INSTANCE_2) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
"! gcloud storage cat $gcs_input_uri" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1632,8 +1626,7 @@
|
||||
"source": [
|
||||
"def get_latest_predictions(gcs_out_dir):\n",
|
||||
" \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gsutil ls $gcs_out_dir\n",
|
||||
" latest = \"\"\n",
|
||||
" folders = !gcloud storage ls $gcs_out_dir\n", " latest = \"\"\n",
|
||||
" for folder in folders:\n",
|
||||
" subfolder = folder.split(\"/\")[-2]\n",
|
||||
" if subfolder.startswith(\"prediction-\"):\n",
|
||||
@@ -1650,10 +1643,8 @@
|
||||
" raise Exception(\"Batch Job Failed\")\n",
|
||||
" else:\n",
|
||||
" folder = get_latest_predictions(predictions)\n",
|
||||
" ! gsutil ls $folder/prediction*.csv\n",
|
||||
"\n",
|
||||
" ! gsutil cat $folder/prediction*.csv\n",
|
||||
" break\n",
|
||||
" ! gcloud storage ls $folder/prediction*.csv\n", "\n",
|
||||
" ! gcloud storage cat $folder/prediction*.csv\n", " break\n",
|
||||
" time.sleep(60)"
|
||||
]
|
||||
},
|
||||
@@ -1747,8 +1738,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+12
-22
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch_explain.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch_explain.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -421,8 +421,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -441,8 +440,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -810,14 +808,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n", "\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n", "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
" raise Exception(\"label column missing\")"
|
||||
@@ -1363,8 +1358,7 @@
|
||||
" f.write(str(INSTANCE_2) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
"! gcloud storage cat $gcs_input_uri" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1631,8 +1625,7 @@
|
||||
"source": [
|
||||
"def get_latest_predictions(gcs_out_dir):\n",
|
||||
" \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n",
|
||||
" folders = !gsutil ls $gcs_out_dir\n",
|
||||
" latest = \"\"\n",
|
||||
" folders = !gcloud storage ls $gcs_out_dir\n", " latest = \"\"\n",
|
||||
" for folder in folders:\n",
|
||||
" subfolder = folder.split(\"/\")[-2]\n",
|
||||
" if subfolder.startswith(\"prediction-\"):\n",
|
||||
@@ -1649,10 +1642,8 @@
|
||||
" raise Exception(\"Batch Job Failed\")\n",
|
||||
" else:\n",
|
||||
" folder = get_latest_predictions(predictions)\n",
|
||||
" ! gsutil ls $folder/explanation*.csv\n",
|
||||
"\n",
|
||||
" ! gsutil cat $folder/explanation*.csv\n",
|
||||
" break\n",
|
||||
" ! gcloud storage ls $folder/explanation*.csv\n", "\n",
|
||||
" ! gcloud storage cat $folder/explanation*.csv\n", " break\n",
|
||||
" time.sleep(60)"
|
||||
]
|
||||
},
|
||||
@@ -1746,8 +1737,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+12
-22
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_export_cloud.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_export_cloud.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -416,8 +416,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage buckets create --location $REGION $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -436,8 +435,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
"! gcloud storage ls --all-versions --long $BUCKET_NAME" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -749,14 +747,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n", "\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n", "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
" raise Exception(\"label column missing\")"
|
||||
@@ -1283,15 +1278,12 @@
|
||||
"source": [
|
||||
"print(\"Model Package:\", model_package)\n",
|
||||
"print(\"Contents:\")\n",
|
||||
"! gsutil ls $model_package\n",
|
||||
"\n",
|
||||
"! gcloud storage ls $model_package\n", "\n",
|
||||
"print(\"\\nTF Saved Model\")\n",
|
||||
"path = model_package + \"/predict\"\n",
|
||||
"files = ! gsutil ls $path\n",
|
||||
"saved_dir = files[1]\n",
|
||||
"files = ! gcloud storage ls $path\n", "saved_dir = files[1]\n",
|
||||
"print(saved_dir)\n",
|
||||
"! gsutil ls $saved_dir"
|
||||
]
|
||||
"! gcloud storage ls $saved_dir" ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1312,8 +1304,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil cp -r $model_package ."
|
||||
]
|
||||
"! gcloud storage cp --recursive $model_package ." ]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -1599,8 +1590,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+6
-10
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -735,14 +735,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n", "print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n", "\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n", "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
" raise Exception(\"label column missing\")"
|
||||
@@ -1645,8 +1642,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
]
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME" ]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
|
||||
+6
-7
@@ -34,12 +34,12 @@
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online_explain.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" <img src=\"https://www.gstatic.com/pantheon/images/bigquery/welcome_page/colab-logo.svg\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online_explain.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img width=\"32px\" src=\"https://www.svgrepo.com/download/217753/github.svg\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -397,7 +397,6 @@
|
||||
"\n",
|
||||
"import google.cloud.aiplatform_v1beta1 as aip\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.json_format import MessageToJson, ParseDict\n",
|
||||
"from google.protobuf.struct_pb2 import Struct, Value"
|
||||
]
|
||||
},
|
||||
@@ -735,13 +734,13 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"count = ! gsutil cat $IMPORT_FILE | wc -l\n",
|
||||
"count = ! gcloud storage cat $IMPORT_FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
"\n",
|
||||
"print(\"First 10 rows\")\n",
|
||||
"! gsutil cat $IMPORT_FILE | head\n",
|
||||
"! gcloud storage cat $IMPORT_FILE | head\n",
|
||||
"\n",
|
||||
"heading = ! gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"heading = ! gcloud storage cat $IMPORT_FILE | head -n1\n",
|
||||
"label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n",
|
||||
"print(\"Label Column Name\", label_column)\n",
|
||||
"if label_column is None:\n",
|
||||
@@ -1820,7 +1819,7 @@
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
" ! gcloud storage rm --recursive $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user