Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f2371b4f7d | ||
|
|
20cb46cc29 | ||
|
|
db1827cb74 | ||
|
|
0137cd106e | ||
|
|
8b3b63d714 | ||
|
|
bf79916f29 | ||
|
|
6bf462f79d | ||
|
|
106cdee495 | ||
|
|
b702bf3a9e | ||
|
|
36ea560aad | ||
|
|
48e744d004 | ||
|
|
dfb7301733 | ||
|
|
da707b2cbc | ||
|
|
873ba9dde9 | ||
|
|
15c452f39d | ||
|
|
be7111815b | ||
|
|
17db1a952b | ||
|
|
455143d0f0 | ||
|
|
f0892852cb | ||
|
|
ca48556d0c | ||
|
|
f961aa3174 | ||
|
|
64c8eca7df | ||
|
|
b3332c1742 | ||
|
|
b35f697a75 | ||
|
|
3bc32a1d48 | ||
|
|
d66f851554 | ||
|
|
d733f107e1 | ||
|
|
805e2e1c83 | ||
|
|
03dc17d9d7 | ||
|
|
a4909f823e | ||
|
|
e91b259595 | ||
|
|
14284046e4 | ||
|
|
59a9a5e6ba | ||
|
|
f3b0a9e0c0 | ||
|
|
92b572a364 | ||
|
|
014be9b530 | ||
|
|
fa675e0082 | ||
|
|
113cbb9709 | ||
|
|
b72bdc8112 | ||
|
|
0842fa8354 | ||
|
|
4e4f3f4095 | ||
|
|
d014febeb9 | ||
|
|
a3264df643 | ||
|
|
f0208e3e37 | ||
|
|
f84e1b9cbc | ||
|
|
16b2086031 | ||
|
|
0e24ba565d | ||
|
|
49718b02a5 | ||
|
|
63031ed364 | ||
|
|
9fd325e25b | ||
|
|
93af43419c | ||
|
|
c8f0cdb74a | ||
|
|
19368fe5cc | ||
|
|
6a05e0eb6c | ||
|
|
40d643f11c | ||
|
|
72009cd21b | ||
|
|
cc9a50f40b | ||
|
|
e3135e8875 | ||
|
|
59eb297151 | ||
|
|
32b0c1c89e | ||
|
|
3328a8190d | ||
|
|
7aa6acdca6 | ||
|
|
7926ff1264 | ||
|
|
67b926dfb4 | ||
|
|
327f9e7a4b | ||
|
|
8be220d089 | ||
|
|
2dad40f49f | ||
|
|
99b724028f | ||
|
|
909fbcb0d4 | ||
|
|
351fc3e4d3 | ||
|
|
252b3d31a3 | ||
|
|
bfdfaab38c | ||
|
|
21a5963f84 | ||
|
|
9452249dce | ||
|
|
49a9df058f | ||
|
|
f99b7e5d17 | ||
|
|
ac03c57a94 | ||
|
|
2e4cf648c5 | ||
|
|
f000baa328 | ||
|
|
113f89604a | ||
|
|
5019a004ce | ||
|
|
a577f3844a | ||
|
|
88c7f0f690 | ||
|
|
a18792499a | ||
|
|
0b5fc8bb3c | ||
|
|
1f77410fda | ||
|
|
45fb57f29c | ||
|
|
3380b394eb | ||
|
|
1286cc5044 | ||
|
|
1ce1af791c | ||
|
|
2587e329ee | ||
|
|
37d4816051 | ||
|
|
d9dff882b8 | ||
|
|
26cc2c5278 | ||
|
|
69aff1bdc4 | ||
|
|
567994f2b1 | ||
|
|
e316c7b8aa | ||
|
|
c07a059a8b | ||
|
|
7b4bbffc41 | ||
|
|
bdc011ef34 | ||
|
|
44ea0d7c61 | ||
|
|
aa950e5ee4 | ||
|
|
247906e50e | ||
|
|
81b2493a44 | ||
|
|
97b0edb222 | ||
|
|
f0a9e4b9fe | ||
|
|
f35bbcaec5 | ||
|
|
2822061dc9 | ||
|
|
be481c8d17 | ||
|
|
605a972122 | ||
|
|
66d98d9fe7 | ||
|
|
6445ed37c9 | ||
|
|
ca745aeee8 | ||
|
|
f806b4927b | ||
|
|
2942eb5d7f | ||
|
|
2a5d6650ee | ||
|
|
43e971f57a | ||
|
|
785779613b | ||
|
|
481193f0ca | ||
|
|
2cb3cccd14 | ||
|
|
7c16766994 | ||
|
|
448d18deca | ||
|
|
97022b0733 | ||
|
|
c2a3e2d4cd | ||
|
|
b2dfcf17c8 | ||
|
|
d061f09281 | ||
|
|
daa64efd40 | ||
|
|
a37deabe27 | ||
|
|
6009ef0def | ||
|
|
edc644b0d0 | ||
|
|
1297af8baf | ||
|
|
b8b1b6675b | ||
|
|
7d3b7abc44 | ||
|
|
9a572f298e | ||
|
|
b53ca9e678 | ||
|
|
9aceec161a | ||
|
|
98b186ed55 | ||
|
|
f23ee1b5a8 | ||
|
|
c6d779f1fc | ||
|
|
8908b27b08 | ||
|
|
8947c9b116 | ||
|
|
9464caac6e | ||
|
|
98ce91c575 | ||
|
|
07f8feda3d | ||
|
|
a4e0496ff5 | ||
|
|
cf162c02c8 | ||
|
|
831aae94df | ||
|
|
3fd9f28778 | ||
|
|
a656e8e2a8 | ||
|
|
2f02152703 | ||
|
|
9d08f8ce67 | ||
|
|
ec6d508793 | ||
|
|
772e35ea75 | ||
|
|
1d28f886c8 | ||
|
|
d3dc8aeb9a | ||
|
|
a07d762934 | ||
|
|
85ac9e127d | ||
|
|
011c2823ff | ||
|
|
f28a94f03f | ||
|
|
5ecfc80cb9 | ||
|
|
e5e36ba050 | ||
|
|
0ad9116d6a | ||
|
|
0516032443 | ||
|
|
95d211c90f | ||
|
|
12d6a75ef7 | ||
|
|
06153dc373 | ||
|
|
02afa91fc3 | ||
|
|
c8b212789f | ||
|
|
95256d3fcf | ||
|
|
8a6d174c99 | ||
|
|
d36cf7f662 | ||
|
|
1b02a542c8 | ||
|
|
45430bb010 | ||
|
|
be2a139ade | ||
|
|
70d77b24f6 | ||
|
|
50c25d6d7b | ||
|
|
5bcdc0bc64 | ||
|
|
eff0f95b58 | ||
|
|
50b53d31bd | ||
|
|
b5a56852f3 | ||
|
|
b7486e34ad | ||
|
|
3edc5f1425 | ||
|
|
65b4b73cb5 | ||
|
|
186c08e8c3 | ||
|
|
7721aa0def | ||
|
|
4987c60e03 | ||
|
|
cd845f7fdd | ||
|
|
9e84d9e782 | ||
|
|
23c7fcc97f | ||
|
|
e4024efbc7 | ||
|
|
c4d53108af | ||
|
|
be8fe3564d | ||
|
|
1f95775057 | ||
|
|
f2a4dd875e | ||
|
|
67dd2300c8 | ||
|
|
b5391b06b4 | ||
|
|
54f2c71c13 | ||
|
|
6697900126 | ||
|
|
9ff3400b44 | ||
|
|
3ff0726ebf | ||
|
|
c96c939dfe | ||
|
|
719cf280c9 | ||
|
|
4ec6e2df04 | ||
|
|
031a9190c3 | ||
|
|
5204dcf327 | ||
|
|
cad623ef84 | ||
|
|
a59f58f8b6 | ||
|
|
d89c613f5d | ||
|
|
fa265ddb2f | ||
|
|
ec3dd04935 | ||
|
|
0c83e81410 | ||
|
|
7808a843cc | ||
|
|
615d7706af | ||
|
|
f05ca4d06a | ||
|
|
cb4145e2b6 | ||
|
|
6917c9aa7b | ||
|
|
c1d2451656 | ||
|
|
c41ec1fabd | ||
|
|
273c91882e | ||
|
|
ad6b5e5830 | ||
|
|
d441eb9d7a | ||
|
|
139d805c9f | ||
|
|
783347fc8e | ||
|
|
6d722d081d | ||
|
|
33a8c6ca0e | ||
|
|
ae043400f5 | ||
|
|
e41f96b31f | ||
|
|
7220f15158 | ||
|
|
b8d7eaa767 | ||
|
|
a612b463a8 | ||
|
|
493e7e81b5 | ||
|
|
b6cf0dbefd | ||
|
|
5d5c08f9e7 | ||
|
|
bfdfac0f81 | ||
|
|
9c72b7e3e7 | ||
|
|
46519e5c64 | ||
|
|
0ff961e203 | ||
|
|
cbc17c6832 | ||
|
|
93e5b15cba | ||
|
|
081e65d076 | ||
|
|
4ab2cfb713 | ||
|
|
a73335c0af | ||
|
|
0f3e257773 | ||
|
|
57d734d84f | ||
|
|
07f2e9c999 | ||
|
|
401883cae3 | ||
|
|
7bc92e1e3c | ||
|
|
de5f8b0653 | ||
|
|
cfa73ed53b | ||
|
|
67370bb1c7 | ||
|
|
f60593255a | ||
|
|
c6f9b97615 | ||
|
|
6161a394c2 | ||
|
|
b35cd42015 | ||
|
|
5e323993db | ||
|
|
f1623e419e | ||
|
|
a3047fb1bb | ||
|
|
b4d02f486e | ||
|
|
9e24893b9b | ||
|
|
47dec6ecef | ||
|
|
059fea672c | ||
|
|
389e804426 | ||
|
|
087a638c18 | ||
|
|
97757c74ca | ||
|
|
08f3ad583b | ||
|
|
16f01d31d3 | ||
|
|
e0f6c66351 | ||
|
|
3733b28772 | ||
|
|
24f8912134 | ||
|
|
cdc8847f9b | ||
|
|
25bd4b9eb5 | ||
|
|
d476191252 | ||
|
|
bf35d6a07c | ||
|
|
5378d38a05 | ||
|
|
85984f1173 | ||
|
|
92bb40349f | ||
|
|
fcee9bf738 | ||
|
|
cc2f3408ad | ||
|
|
89fb218041 | ||
|
|
4d0c3781e5 | ||
|
|
e2e319ac7f | ||
|
|
63508cad3f | ||
|
|
0b6eb6eed1 | ||
|
|
a41cfdeba5 | ||
|
|
7a6763ca33 | ||
|
|
2a6e19e4f9 | ||
|
|
9f9526b722 | ||
|
|
267684b7c3 | ||
|
|
b74dd3d576 | ||
|
|
278ae48842 | ||
|
|
189ca12627 | ||
|
|
5108f57ba8 |
@@ -1,42 +1,45 @@
|
||||
from typing import List
|
||||
from resource_cleanup_manager import (
|
||||
ResourceCleanupManager,
|
||||
DatasetResourceCleanupManager,
|
||||
EndpointResourceCleanupManager,
|
||||
ModelResourceCleanupManager,
|
||||
ResourceCleanupManager,
|
||||
DatasetResourceCleanupManager,
|
||||
EndpointResourceCleanupManager,
|
||||
ModelResourceCleanupManager,
|
||||
)
|
||||
|
||||
|
||||
def run_cleanup_managers(managers: List[ResourceCleanupManager], is_dry_run: bool):
|
||||
for manager in managers:
|
||||
type_name = manager.type_name
|
||||
for manager in managers:
|
||||
type_name = manager.type_name
|
||||
|
||||
print(f"Fetching {type_name}'s...")
|
||||
resources = manager.list()
|
||||
print(f"Found {len(resources)} {type_name}'s")
|
||||
for resource in resources:
|
||||
if not manager.is_deletable(resource):
|
||||
continue
|
||||
print(f"Fetching {type_name}'s...")
|
||||
resources = manager.list()
|
||||
print(f"Found {len(resources)} {type_name}'s")
|
||||
for resource in resources:
|
||||
if not manager.is_deletable(resource):
|
||||
continue
|
||||
|
||||
if is_dry_run:
|
||||
resource_name = manager.resource_name(resource)
|
||||
print(f"Will delete '{type_name}': {resource_name}")
|
||||
else:
|
||||
manager.delete(resource)
|
||||
if is_dry_run:
|
||||
resource_name = manager.resource_name(resource)
|
||||
print(f"Will delete '{type_name}': {resource_name}")
|
||||
else:
|
||||
try:
|
||||
manager.delete(resource)
|
||||
except Exception as exception:
|
||||
print(exception)
|
||||
|
||||
print("")
|
||||
print("")
|
||||
|
||||
|
||||
is_dry_run = False
|
||||
|
||||
if is_dry_run:
|
||||
print("Starting cleanup in dry run mode...")
|
||||
print("Starting cleanup in dry run mode...")
|
||||
|
||||
# List of all cleanup managers
|
||||
managers = [
|
||||
DatasetResourceCleanupManager(),
|
||||
EndpointResourceCleanupManager(),
|
||||
ModelResourceCleanupManager(),
|
||||
DatasetResourceCleanupManager(),
|
||||
EndpointResourceCleanupManager(),
|
||||
ModelResourceCleanupManager(),
|
||||
]
|
||||
|
||||
run_cleanup_managers(managers=managers, is_dry_run=is_dry_run)
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
#!/usr/bin/env python
|
||||
# Copyright 2021 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""A CLI to process changed notebooks and execute them on Google Cloud Build"""
|
||||
|
||||
import argparse
|
||||
import pathlib
|
||||
import execute_changed_notebooks_helper
|
||||
|
||||
|
||||
def str2bool(v):
|
||||
if isinstance(v, bool):
|
||||
return v
|
||||
if v.lower() in ("yes", "true", "t", "y", "1"):
|
||||
return True
|
||||
elif v.lower() in ("no", "false", "f", "n", "0"):
|
||||
return False
|
||||
else:
|
||||
raise argparse.ArgumentTypeError("Boolean value expected.")
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
parser.add_argument(
|
||||
"--test_paths_file",
|
||||
type=pathlib.Path,
|
||||
help="The path to the file that has newline-limited folders of notebooks that should be tested.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--base_branch",
|
||||
help="The base git branch to diff against to find changed files.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container_uri",
|
||||
type=str,
|
||||
help="The container uri to run each notebook in.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_project_id",
|
||||
type=str,
|
||||
help="The GCP project id. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_region",
|
||||
type=str,
|
||||
help="The GCP region. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--staging_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for staging temporary files.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifacts_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for storing executed notebooks.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--private_pool_id",
|
||||
type=str,
|
||||
help="The private pool id.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--should_parallelize",
|
||||
type=str2bool,
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=True,
|
||||
help="Should run notebooks in parallel.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
notebooks = execute_changed_notebooks_helper.get_changed_notebooks(
|
||||
test_paths_file=args.test_paths_file,
|
||||
base_branch=args.base_branch,
|
||||
)
|
||||
|
||||
execute_changed_notebooks_helper.process_and_execute_notebooks(
|
||||
notebooks=notebooks,
|
||||
container_uri=args.container_uri,
|
||||
staging_bucket=args.staging_bucket,
|
||||
artifacts_bucket=args.artifacts_bucket,
|
||||
variable_project_id=args.variable_project_id,
|
||||
variable_region=args.variable_region,
|
||||
private_pool_id=args.private_pool_id if not "default" else None,
|
||||
should_parallelize=args.should_parallelize,
|
||||
)
|
||||
@@ -13,7 +13,6 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import concurrent
|
||||
import dataclasses
|
||||
import datetime
|
||||
@@ -32,17 +31,6 @@ from utils import util, NotebookProcessors
|
||||
from google.cloud.devtools.cloudbuild_v1.types import BuildOperationMetadata
|
||||
|
||||
|
||||
def str2bool(v):
|
||||
if isinstance(v, bool):
|
||||
return v
|
||||
if v.lower() in ("yes", "true", "t", "y", "1"):
|
||||
return True
|
||||
elif v.lower() in ("no", "false", "f", "n", "0"):
|
||||
return False
|
||||
else:
|
||||
raise argparse.ArgumentTypeError("Boolean value expected.")
|
||||
|
||||
|
||||
def format_timedelta(delta: datetime.timedelta) -> str:
|
||||
"""Formats a timedelta duration to [N days] %H:%M:%S format"""
|
||||
seconds = int(delta.total_seconds())
|
||||
@@ -115,12 +103,13 @@ def _create_tag(filepath: str) -> str:
|
||||
return tag
|
||||
|
||||
|
||||
def execute_notebook(
|
||||
def process_and_execute_notebook(
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
private_pool_id: Optional[str],
|
||||
notebook: str,
|
||||
should_get_tail_logs: bool = False,
|
||||
) -> NotebookExecutionResult:
|
||||
@@ -162,6 +151,8 @@ def execute_notebook(
|
||||
notebook_output_uri=notebook_output_uri,
|
||||
container_uri=container_uri,
|
||||
tag=tag,
|
||||
region=variable_region,
|
||||
private_pool_id=private_pool_id,
|
||||
)
|
||||
|
||||
operation_metadata = BuildOperationMetadata(mapping=operation.metadata)
|
||||
@@ -202,41 +193,13 @@ def execute_notebook(
|
||||
return result
|
||||
|
||||
|
||||
def run_changed_notebooks(
|
||||
def get_changed_notebooks(
|
||||
test_paths_file: str,
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
should_parallelize: bool,
|
||||
base_branch: Optional[str] = None,
|
||||
):
|
||||
) -> List[str]:
|
||||
"""
|
||||
Run the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only runs notebooks that have differences from the Git base_branch.
|
||||
|
||||
The executed notebooks are saved in the artifacts_bucket.
|
||||
|
||||
Variables are also injected into the notebooks such as the variable_project_id and variable_region.
|
||||
|
||||
Args:
|
||||
test_paths_file (str):
|
||||
Required. The new-line delimited file to folders and files that need checking.
|
||||
Folders are checked recursively.
|
||||
base_branch (str):
|
||||
Optional. If provided, only the files that have changed from the base_branch will be checked.
|
||||
If not provided, all files will be checked.
|
||||
staging_bucket (str):
|
||||
Required. The GCS staging bucket to write source code to.
|
||||
artifacts_bucket (str):
|
||||
Required. The GCS staging bucket to write executed notebooks to.
|
||||
variable_project_id (str):
|
||||
Required. The value for PROJECT_ID to inject into notebooks.
|
||||
variable_region (str):
|
||||
Required. The value for REGION to inject into notebooks.
|
||||
should_parallelize (bool):
|
||||
Required. Should run notebooks in parallel using a thread pool as opposed to in sequence.
|
||||
Get the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only returns notebooks that have differences from the Git base_branch.
|
||||
"""
|
||||
|
||||
test_paths = []
|
||||
@@ -266,6 +229,45 @@ def run_changed_notebooks(
|
||||
notebooks = [notebook for notebook in notebooks if len(notebook) > 0]
|
||||
notebooks = [notebook for notebook in notebooks if pathlib.Path(notebook).exists()]
|
||||
|
||||
return notebooks
|
||||
|
||||
|
||||
def process_and_execute_notebooks(
|
||||
notebooks: List[str],
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
private_pool_id: Optional[str],
|
||||
should_parallelize: bool,
|
||||
):
|
||||
"""
|
||||
Run the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only runs notebooks that have differences from the Git base_branch.
|
||||
|
||||
The executed notebooks are saved in the artifacts_bucket.
|
||||
|
||||
Variables are also injected into the notebooks such as the variable_project_id and variable_region.
|
||||
|
||||
Args:
|
||||
test_paths_file (str):
|
||||
Required. The new-line delimited file to folders and files that need checking.
|
||||
Folders are checked recursively.
|
||||
base_branch (str):
|
||||
Optional. If provided, only the files that have changed from the base_branch will be checked.
|
||||
If not provided, all files will be checked.
|
||||
staging_bucket (str):
|
||||
Required. The GCS staging bucket to write source code to.
|
||||
artifacts_bucket (str):
|
||||
Required. The GCS staging bucket to write executed notebooks to.
|
||||
variable_project_id (str):
|
||||
Required. The value for PROJECT_ID to inject into notebooks.
|
||||
variable_region (str):
|
||||
Required. The value for REGION to inject into notebooks.
|
||||
should_parallelize (bool):
|
||||
Required. Should run notebooks in parallel using a thread pool as opposed to in sequence.
|
||||
"""
|
||||
notebook_execution_results: List[NotebookExecutionResult] = []
|
||||
|
||||
if len(notebooks) > 0:
|
||||
@@ -279,24 +281,26 @@ def run_changed_notebooks(
|
||||
notebook_execution_results = list(
|
||||
executor.map(
|
||||
functools.partial(
|
||||
execute_notebook,
|
||||
process_and_execute_notebook,
|
||||
container_uri,
|
||||
staging_bucket,
|
||||
artifacts_bucket,
|
||||
variable_project_id,
|
||||
variable_region,
|
||||
private_pool_id,
|
||||
),
|
||||
notebooks,
|
||||
)
|
||||
)
|
||||
else:
|
||||
notebook_execution_results = [
|
||||
execute_notebook(
|
||||
process_and_execute_notebook(
|
||||
container_uri=container_uri,
|
||||
staging_bucket=staging_bucket,
|
||||
artifacts_bucket=artifacts_bucket,
|
||||
variable_project_id=variable_project_id,
|
||||
variable_region=variable_region,
|
||||
private_pool_id=private_pool_id,
|
||||
notebook=notebook,
|
||||
)
|
||||
for notebook in notebooks
|
||||
@@ -341,67 +345,3 @@ def run_changed_notebooks(
|
||||
# Raise error if any notebooks failed
|
||||
if not all([result.is_pass for result in results_sorted]):
|
||||
raise RuntimeError("Notebook failures detected. See logs for details")
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
parser.add_argument(
|
||||
"--test_paths_file",
|
||||
type=pathlib.Path,
|
||||
help="The path to the file that has newline-limited folders of notebooks that should be tested.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--base_branch",
|
||||
help="The base git branch to diff against to find changed files.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container_uri",
|
||||
type=str,
|
||||
help="The container uri to run each notebook in.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_project_id",
|
||||
type=str,
|
||||
help="The GCP project id. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_region",
|
||||
type=str,
|
||||
help="The GCP region. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--staging_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for staging temporary files.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifacts_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for storing executed notebooks.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--should_parallelize",
|
||||
type=str2bool,
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=True,
|
||||
help="Should run notebooks in parallel.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
run_changed_notebooks(
|
||||
test_paths_file=args.test_paths_file,
|
||||
container_uri=args.container_uri,
|
||||
staging_bucket=args.staging_bucket,
|
||||
artifacts_bucket=args.artifacts_bucket,
|
||||
variable_project_id=args.variable_project_id,
|
||||
variable_region=args.variable_region,
|
||||
should_parallelize=args.should_parallelize,
|
||||
base_branch=args.base_branch,
|
||||
)
|
||||
@@ -13,10 +13,12 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import ExecuteNotebook
|
||||
"""A CLI to download (optional) and run a single notebook locally"""
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
import argparse
|
||||
import execute_notebook_helper
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run a single notebook locally.")
|
||||
parser.add_argument(
|
||||
"--notebook_source",
|
||||
type=str,
|
||||
@@ -31,7 +33,7 @@ parser.add_argument(
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
ExecuteNotebook.execute_notebook(
|
||||
execute_notebook_helper.execute_notebook(
|
||||
notebook_source=args.notebook_source,
|
||||
output_file_or_uri=args.output_file_or_uri,
|
||||
should_log_output=True,
|
||||
|
||||
@@ -13,6 +13,8 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Methods to run a notebook locally"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import errno
|
||||
@@ -30,6 +32,7 @@ def execute_notebook(
|
||||
output_file_or_uri: str,
|
||||
should_log_output: bool,
|
||||
):
|
||||
"""Execute a single notebook using Papermill"""
|
||||
file_name = os.path.basename(os.path.normpath(notebook_source))
|
||||
|
||||
# Download notebook if it's a GCS URI
|
||||
@@ -1,3 +1,21 @@
|
||||
#!/usr/bin/env python
|
||||
# Copyright 2021 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Methods to run a notebook on Google Cloud Build"""
|
||||
|
||||
from re import sub
|
||||
from google.protobuf import duration_pb2
|
||||
from yaml.loader import FullLoader
|
||||
|
||||
@@ -9,10 +27,12 @@ from typing import Optional
|
||||
import yaml
|
||||
|
||||
from google.cloud.aiplatform import utils
|
||||
from google.api_core import operation
|
||||
from google.api_core import operation, client_options
|
||||
|
||||
|
||||
CLOUD_BUILD_FILEPATH = ".cloud-build/notebook-execution-test-cloudbuild-single.yaml"
|
||||
TIMEOUT_IN_SECONDS = 86400
|
||||
SERVICE_BASE_PATH = "cloudbuild.googleapis.com"
|
||||
|
||||
|
||||
def execute_notebook_remote(
|
||||
@@ -20,20 +40,12 @@ def execute_notebook_remote(
|
||||
notebook_uri: str,
|
||||
notebook_output_uri: str,
|
||||
container_uri: str,
|
||||
region: str,
|
||||
private_pool_id: Optional[str],
|
||||
tag: Optional[str],
|
||||
) -> operation.Operation:
|
||||
"""Create and execute a simple Google Cloud Build configuration,
|
||||
print the in-progress status and print the completed status."""
|
||||
|
||||
# Authorize the client with Google defaults
|
||||
credentials, project_id = google.auth.default()
|
||||
client = cloudbuild_v1.services.cloud_build.CloudBuildClient()
|
||||
|
||||
build = cloudbuild_v1.Build()
|
||||
|
||||
# The following build steps will output "hello world"
|
||||
# For more information on build configuration, see
|
||||
# https://cloud.google.com/build/docs/configuring-builds/create-basic-configuration
|
||||
"""Create and execute a single notebook on Google Cloud Build"""
|
||||
# Load build steps from YAML
|
||||
cloudbuild_config = yaml.load(open(CLOUD_BUILD_FILEPATH), Loader=FullLoader)
|
||||
|
||||
substitutions = {
|
||||
@@ -42,6 +54,23 @@ def execute_notebook_remote(
|
||||
"_NOTEBOOK_OUTPUT_GCS_URI": notebook_output_uri,
|
||||
}
|
||||
|
||||
build = cloudbuild_v1.Build()
|
||||
|
||||
options: Optional[client_options.ClientOptions] = None
|
||||
if private_pool_id:
|
||||
substitutions["_PRIVATE_POOL_NAME"] = private_pool_id
|
||||
build.options = cloudbuild_config["options"]
|
||||
|
||||
# Switch to the regional endpoint of the pool
|
||||
options = client_options.ClientOptions(
|
||||
api_endpoint=f"{region}-{SERVICE_BASE_PATH}"
|
||||
)
|
||||
|
||||
# Authorize the client with Google defaults
|
||||
credentials, project_id = google.auth.default()
|
||||
|
||||
client = cloudbuild_v1.services.cloud_build.CloudBuildClient(client_options=options)
|
||||
|
||||
(
|
||||
source_archived_file_gcs_bucket,
|
||||
source_archived_file_gcs_object,
|
||||
|
||||
@@ -26,3 +26,6 @@ steps:
|
||||
env:
|
||||
- 'IS_TESTING=1'
|
||||
timeout: 86400s
|
||||
options:
|
||||
pool:
|
||||
name: ${_PRIVATE_POOL_NAME}
|
||||
@@ -5,10 +5,6 @@ steps:
|
||||
args:
|
||||
- -c
|
||||
- 'gcloud config list'
|
||||
# # Clone the Git repo
|
||||
# - name: ${_PYTHON_IMAGE}
|
||||
# entrypoint: git
|
||||
# args: ['clone', "${_GIT_REPO}", "--branch", "${_GIT_BRANCH_NAME}", "."]
|
||||
# Check the Python version
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
@@ -28,11 +24,15 @@ steps:
|
||||
- -c
|
||||
- 'python3 -m pip install -U pip && python3 -m pip install -U --user -r .cloud-build/requirements.txt'
|
||||
# Install Python dependencies and run testing script
|
||||
# TODO: Only pass in private_pool_id if it is set
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
args:
|
||||
- -c
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/ExecuteChangedNotebooks.py --test_paths_file "${_TEST_PATHS_FILE}" --base_branch "${_FORCED_BASE_BRANCH}" --container_uri ${_PYTHON_IMAGE} --staging_bucket ${_GCS_STAGING_BUCKET} --artifacts_bucket ${_GCS_STAGING_BUCKET}/executed_notebooks/PR_${_PR_NUMBER}/BUILD_${BUILD_ID} --variable_project_id ${PROJECT_ID} --variable_region ${_GCP_REGION}'
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/execute_changed_notebooks_cli.py --test_paths_file "${_TEST_PATHS_FILE}" --base_branch "${_FORCED_BASE_BRANCH}" --container_uri ${_PYTHON_IMAGE} --staging_bucket ${_GCS_STAGING_BUCKET} --artifacts_bucket ${_GCS_STAGING_BUCKET}/executed_notebooks/PR_${_PR_NUMBER}/BUILD_${BUILD_ID} --variable_project_id ${PROJECT_ID} --variable_region ${_GCP_REGION} `if [ ! -z "${_PRIVATE_POOL_NAME}" ]; then echo "--private_pool_id ${_PRIVATE_POOL_NAME}"; fi`'
|
||||
env:
|
||||
- 'IS_TESTING=1'
|
||||
timeout: 86400s
|
||||
options:
|
||||
pool:
|
||||
name: ${_PRIVATE_POOL_NAME}
|
||||
@@ -1,10 +1,10 @@
|
||||
ipython==8.0.0
|
||||
jupyter==1.0.0
|
||||
nbconvert==6.4.0
|
||||
papermill==2.3.3
|
||||
numpy==1.22.0
|
||||
pandas==1.3.5
|
||||
matplotlib==3.5.1
|
||||
ipython
|
||||
numpy
|
||||
jupyter
|
||||
nbconvert
|
||||
papermill
|
||||
pandas
|
||||
matplotlib
|
||||
tabulate
|
||||
google-cloud-aiplatform
|
||||
google-cloud-storage
|
||||
|
||||
@@ -1,2 +1,3 @@
|
||||
notebooks/official
|
||||
notebooks/notebook_template.ipynb
|
||||
notebooks/community/ml_ops
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
notebooks/official/vizier/gapic-vizier-multi-objective-optimization.ipynb
|
||||
notebooks/official/pipelines/lightweight_functions_component_io_kfp.ipynb
|
||||
notebooks/official/matching_engine/intro-swivel.ipynb
|
||||
notebooks/official/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb
|
||||
notebooks/official/pipelines/metrics_viz_run_compare_kfp.ipynb
|
||||
@@ -11,7 +11,7 @@ import uuid
|
||||
|
||||
|
||||
def download_file(bucket_name: str, blob_name: str, destination_file: str) -> str:
|
||||
"""Copies a remote GCS file to a local path."""
|
||||
"""Copies a remote GCS file to a local path"""
|
||||
remote_file_path = "".join(["gs://", "/".join([bucket_name, blob_name])])
|
||||
|
||||
subprocess.check_output(
|
||||
@@ -25,7 +25,7 @@ def upload_file(
|
||||
local_file_path: str,
|
||||
remote_file_path: str,
|
||||
) -> str:
|
||||
"""Copies a local file to a GCS path."""
|
||||
"""Copies a local file to a GCS path"""
|
||||
subprocess.check_output(
|
||||
["gsutil", "cp", local_file_path, remote_file_path], encoding="UTF-8"
|
||||
)
|
||||
|
||||
@@ -7,9 +7,9 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v2
|
||||
uses: actions/setup-python@v3
|
||||
- name: Fetch pull request branch
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Fetch base main branch
|
||||
|
||||
@@ -2,8 +2,8 @@ git+https://github.com/tensorflow/docs
|
||||
ipython
|
||||
jupyter
|
||||
nbconvert
|
||||
black==21.10b0
|
||||
pyupgrade==2.29.1
|
||||
black==22.3.0
|
||||
pyupgrade==2.31.1
|
||||
isort==5.10.1
|
||||
flake8==4.0.1
|
||||
nbqa==1.2.2
|
||||
nbqa==1.3.1
|
||||
|
||||
@@ -84,19 +84,19 @@ if [ ${#notebooks[@]} -gt 0 ]; then
|
||||
FLAKE8_RTN=$?
|
||||
else
|
||||
echo "Running black..."
|
||||
python3 -m nbqa black "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa black "$notebook"
|
||||
BLACK_RTN=$?
|
||||
echo "Running pyupgrade..."
|
||||
python3 -m nbqa pyupgrade "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa pyupgrade "$notebook"
|
||||
PYUPGRADE_RTN=$?
|
||||
echo "Running isort..."
|
||||
python3 -m nbqa isort "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa isort "$notebook"
|
||||
ISORT_RTN=$?
|
||||
echo "Running nbfmt..."
|
||||
python3 -m tensorflow_docs.tools.nbfmt --remove_outputs "$notebook"
|
||||
NBFMT_RTN=$?
|
||||
echo "Running flake8..."
|
||||
python3 -m nbqa flake8 "$notebook" --show-source --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291 --nbqa-mutate
|
||||
python3 -m nbqa flake8 "$notebook" --show-source --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291
|
||||
FLAKE8_RTN=$?
|
||||
fi
|
||||
|
||||
|
||||
@@ -6,7 +6,19 @@ Welcome to the Google Cloud [Vertex AI](https://cloud.google.com/vertex-ai/docs/
|
||||
|
||||
## Overview
|
||||
|
||||
The repository contains [Notebooks](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks) and [Community Content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/community-content) that demonstrate how to develop and manage ML workflows using Google Cloud Vertex AI.
|
||||
The repository contains [notebooks](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks) and [community content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/community-content) that demonstrate how to develop and manage ML workflows using Google Cloud Vertex AI.
|
||||
|
||||
## Repository structure
|
||||
|
||||
```bash
|
||||
├── community-content - Sample code and tutorials contributed by the community
|
||||
├── notebooks
|
||||
│ ├── community - Notebooks contributed by the community
|
||||
│ ├── official - Notebooks demonstrating use of each Vertex AI service
|
||||
│ │ ├── automl
|
||||
│ │ ├── custom
|
||||
│ │ ├── ...
|
||||
```
|
||||
|
||||
## Contributing
|
||||
|
||||
@@ -19,3 +31,7 @@ Please use the [issues page](https://github.com/GoogleCloudPlatform/vertex-ai-sa
|
||||
## Disclaimer
|
||||
|
||||
This is not an officially supported Google product. The code in this repository is for demonstrative purposes only.
|
||||
|
||||
## Feedback
|
||||
|
||||
Please feel free to fill out our [survey](https://bit.ly/vertex-ai-samples-survey) to give us feedback on the repo and its content.
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
* @vertex-ai-samples-contributors @GoogleCloudPlatform/cloudml-samples-owners
|
||||
/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk @yinghsienwu
|
||||
/pytorch_text_classification_using_vertex_sdk_and_gcloud @RajeshThallam
|
||||
/pytorch_text_classification_using_vertex_sdk_and_gcloud @RajeshThallam @ultrons
|
||||
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
|
||||
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
|
||||
/pluto_on_workbench @wkharold
|
||||
|
||||
@@ -0,0 +1,824 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "pc5-mbsX9PZC"
|
||||
},
|
||||
"source": [
|
||||
"# AlphaFold On Vertex AI Workbench\n",
|
||||
"\n",
|
||||
"[Vertex AI Workbench](https://cloud.google.com/vertex-ai/docs/workbench) offers an end-to-end notebook-based production environment that can be preconfigured with the runtime dependencies necessary to run AlphaFold on Vertex AI. With [User-Managed Notebooks](https://cloud.google.com/vertex-ai/docs/workbench/user-managed/introduction), you can configure a GPU accelerator to run AlphaFold using Tensorflow, without having to install and manage drivers or JupyterLab instances. This notebook allows you to easily predict the structure of a protein using a slightly simplified version of [AlphaFold v2.1.0](https://doi.org/10.1038/s41586-021-03819-2). \n",
|
||||
"\n",
|
||||
"##  [Launch this Notebook in Vertex AI Workbench](https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/raw/main/community-content/alphafold_on_workbench/AlphaFold.ipynb)\n",
|
||||
"\n",
|
||||
"**Differences to AlphaFold v2.1.0**\n",
|
||||
"\n",
|
||||
"In comparison to AlphaFold v2.1.0, this notebook notebook uses **no templates (homologous structures)** and a selected portion of the [BFD database](https://bfd.mmseqs.com/). We have validated these changes on several thousand recent PDB structures. While accuracy will be near-identical to the full AlphaFold system on many targets, a small fraction have a large drop in accuracy due to the smaller MSA and lack of templates. For best reliability, we recommend instead using the [full open source AlphaFold](https://github.com/deepmind/alphafold/), or the [AlphaFold Protein Structure Database](https://alphafold.ebi.ac.uk/).\n",
|
||||
"\n",
|
||||
"**This notebook has an small drop in average accuracy for multimers compared to local AlphaFold installation, for full multimer accuracy it is highly recommended to run [AlphaFold locally](https://github.com/deepmind/alphafold#running-alphafold).** Moreover, the AlphaFold-Multimer requires searching for MSA for every unique sequence in the complex, hence it is substantially slower. If your notebook times-out due to slow multimer MSA search, we recommend running AlphaFold locally.\n",
|
||||
"\n",
|
||||
"Please note that this notebook is provided as an early-access prototype and is not a finished product. It is provided for theoretical modelling only and caution should be exercised in its use. \n",
|
||||
"\n",
|
||||
"**Citing this work**\n",
|
||||
"\n",
|
||||
"Any publication that discloses findings arising from using this notebook should [cite](https://github.com/deepmind/alphafold/#citing-this-work) the [AlphaFold paper](https://doi.org/10.1038/s41586-021-03819-2).\n",
|
||||
"\n",
|
||||
"**Licenses**\n",
|
||||
"\n",
|
||||
"This Colab uses the [AlphaFold model parameters](https://github.com/deepmind/alphafold/#model-parameters-license) which are subject to the Creative Commons Attribution 4.0 International ([CC BY 4.0](https://creativecommons.org/licenses/by/4.0/legalcode)) license. The Colab itself is provided under the [Apache 2.0 license](https://www.apache.org/licenses/LICENSE-2.0). See the full license statement below.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"**More information**\n",
|
||||
"\n",
|
||||
"You can find more information about how AlphaFold works in the following papers:\n",
|
||||
"\n",
|
||||
"* [AlphaFold methods paper](https://www.nature.com/articles/s41586-021-03819-2)\n",
|
||||
"* [AlphaFold predictions of the human proteome paper](https://www.nature.com/articles/s41586-021-03828-1)\n",
|
||||
"* [AlphaFold-Multimer paper](https://www.biorxiv.org/content/10.1101/2021.10.04.463034v1)\n",
|
||||
"\n",
|
||||
"FAQ on how to interpret AlphaFold predictions are [here](https://alphafold.ebi.ac.uk/faq)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "b7a02613eb1a"
|
||||
},
|
||||
"source": [
|
||||
"## Download AlphaFold Data"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "woIxeCPygt7K"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import subprocess\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"import alphafold.common\n",
|
||||
"import tqdm.notebook\n",
|
||||
"from IPython.utils import io\n",
|
||||
"\n",
|
||||
"TQDM_BAR_FORMAT = (\n",
|
||||
" \"{l_bar}{bar}| {n_fmt}/{total_fmt} [elapsed: {elapsed} remaining: {remaining}]\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"SOURCE_URL = (\n",
|
||||
" \"https://storage.googleapis.com/alphafold/alphafold_params_colab_2022-01-19.tar\"\n",
|
||||
")\n",
|
||||
"PARAMS_DIR = \"alphafold/data/params\"\n",
|
||||
"PARAMS_PATH = os.path.join(PARAMS_DIR, os.path.basename(SOURCE_URL))\n",
|
||||
"ALPHAFOLD_COMMON_DIR = os.path.dirname(alphafold.common.__file__)\n",
|
||||
"\n",
|
||||
"try:\n",
|
||||
" with tqdm.notebook.tqdm(total=100, bar_format=TQDM_BAR_FORMAT) as pbar:\n",
|
||||
" with io.capture_output() as captured:\n",
|
||||
"\n",
|
||||
" # Download and store stereo_chemical_props.txt\n",
|
||||
" !mkdir -p ~/content/alphafold/alphafold/common\n",
|
||||
" !mkdir -p /opt/conda/lib/python3.7/site-packages/alphafold/common/\n",
|
||||
" !wget -q -P ~/content/alphafold/alphafold/common https://git.scicore.unibas.ch/schwede/openstructure/-/raw/7102c63615b64735c4941278d92b554ec94415f8/modules/mol/alg/src/stereo_chemical_props.txt\n",
|
||||
" pbar.update(18)\n",
|
||||
" !cp -f ~/content/alphafold/alphafold/common/stereo_chemical_props.txt \"{ALPHAFOLD_COMMON_DIR}\"\n",
|
||||
"\n",
|
||||
" # Download alphafold_params_colab_2021-10-27.tar\n",
|
||||
" !mkdir --parents \"{PARAMS_DIR}\"\n",
|
||||
" !wget -O \"{PARAMS_PATH}\" \"{SOURCE_URL}\"\n",
|
||||
" pbar.update(27)\n",
|
||||
"\n",
|
||||
" # Un-tar alphafold_params_colab_2021-10-27.tar\n",
|
||||
" !tar --extract --verbose --file=\"{PARAMS_PATH}\" --directory=\"{PARAMS_DIR}\" --preserve-permissions\n",
|
||||
" # !rm \"{PARAMS_PATH}\"\n",
|
||||
" pbar.update(55)\n",
|
||||
"\n",
|
||||
"except subprocess.CalledProcessError:\n",
|
||||
" print(captured)\n",
|
||||
" raise"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d8926b7d5529"
|
||||
},
|
||||
"source": [
|
||||
"## Configure GPU Acceleration"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "VzJ5iMjTtoZw"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Confirm accelerator configuration\n",
|
||||
"import jax\n",
|
||||
"\n",
|
||||
"if jax.local_devices()[0].platform == \"tpu\":\n",
|
||||
" raise RuntimeError(\n",
|
||||
" \"TPU runtime not supported. Please configure GPU acceleration on the VM.\"\n",
|
||||
" )\n",
|
||||
"elif jax.local_devices()[0].platform == \"cpu\":\n",
|
||||
" print(\n",
|
||||
" \"CPU-only runtime is not recommended, because prediction execution will be slow. For better performance, consider GPU acceleration on the VM.\"\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" print(f\"Running with {jax.local_devices()[0].device_kind} GPU\")\n",
|
||||
"\n",
|
||||
"# Make sure all necessary environment variables are set.\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"os.environ[\"TF_FORCE_UNIFIED_MEMORY\"] = \"1\"\n",
|
||||
"os.environ[\"XLA_PYTHON_CLIENT_MEM_FRACTION\"] = \"2.0\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "W4JpOs6oA-QS"
|
||||
},
|
||||
"source": [
|
||||
"## Making a prediction\n",
|
||||
"\n",
|
||||
"Please paste the sequence of your protein in the text box below, then run the remaining cells via _Run_ > _Run Selected Cell and All Below_. You can also run the cells individually by pressing the _Play_ button on the left.\n",
|
||||
"\n",
|
||||
"Note that the search against databases and the actual prediction can take some time, from minutes to hours, depending on the length of the protein and what type of GPU you allocate (see FAQ below).\n",
|
||||
"\n",
|
||||
"To start, enter the amino acid sequence(s) to fold ⬇️\n",
|
||||
"\n",
|
||||
"If you enter only a single sequence, the monomer model will be used. If you enter multiple sequences, the multimer model will be used."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b310d44229d0"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Input sequences (type: str)\n",
|
||||
"sequence_1 = \"MAAHKGAEHHHKAAEHHEQAAKHHHAAAEHHEKGEHEQAAHHADTAYAHHKHAEEHAAQAAKHDAEHHAPKPH\"\n",
|
||||
"sequence_2 = \"\"\n",
|
||||
"sequence_3 = \"\"\n",
|
||||
"sequence_4 = \"\"\n",
|
||||
"sequence_5 = \"\"\n",
|
||||
"sequence_6 = \"\"\n",
|
||||
"sequence_7 = \"\"\n",
|
||||
"sequence_8 = \"\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "rowN0bVYLe9n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from alphafold.notebooks import notebook_utils\n",
|
||||
"\n",
|
||||
"input_sequences = (\n",
|
||||
" sequence_1,\n",
|
||||
" sequence_2,\n",
|
||||
" sequence_3,\n",
|
||||
" sequence_4,\n",
|
||||
" sequence_5,\n",
|
||||
" sequence_6,\n",
|
||||
" sequence_7,\n",
|
||||
" sequence_8,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# If folding a complex target and all the input sequences are\n",
|
||||
"# prokaryotic then set `is_prokaryotic` to `True`. Set to `False`\n",
|
||||
"# otherwise or if the origin is unknown.\n",
|
||||
"\n",
|
||||
"is_prokaryote = False # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"MIN_SINGLE_SEQUENCE_LENGTH = 16\n",
|
||||
"MAX_SINGLE_SEQUENCE_LENGTH = 2500\n",
|
||||
"MAX_MULTIMER_LENGTH = 2500\n",
|
||||
"\n",
|
||||
"# Validate the input.\n",
|
||||
"sequences, model_type_to_use = notebook_utils.validate_input(\n",
|
||||
" input_sequences=input_sequences,\n",
|
||||
" min_length=MIN_SINGLE_SEQUENCE_LENGTH,\n",
|
||||
" max_length=MAX_SINGLE_SEQUENCE_LENGTH,\n",
|
||||
" max_multimer_length=MAX_MULTIMER_LENGTH,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "db551d4877ea"
|
||||
},
|
||||
"source": [
|
||||
"## Search against genetic databases\n",
|
||||
"\n",
|
||||
"Once this cell has been executed, you will see statistics about the multiple sequence alignment (MSA) that will be used by AlphaFold. In particular, you’ll see how well each residue is covered by similar sequences in the MSA."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "2tTeTTsLKPjB"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import collections\n",
|
||||
"import copy\n",
|
||||
"import random\n",
|
||||
"from concurrent import futures\n",
|
||||
"from urllib import request\n",
|
||||
"\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import numpy as np\n",
|
||||
"import py3Dmol\n",
|
||||
"from alphafold.common import protein\n",
|
||||
"from alphafold.data import (feature_processing, msa_pairing, pipeline,\n",
|
||||
" pipeline_multimer)\n",
|
||||
"from alphafold.data.tools import jackhmmer\n",
|
||||
"from alphafold.model import config, data, model\n",
|
||||
"from alphafold.relax import relax, utils\n",
|
||||
"from IPython import display\n",
|
||||
"from ipywidgets import GridspecLayout, Output\n",
|
||||
"\n",
|
||||
"# Color bands for visualizing plddt\n",
|
||||
"PLDDT_BANDS = [\n",
|
||||
" (0, 50, \"#FF7D45\"),\n",
|
||||
" (50, 70, \"#FFDB13\"),\n",
|
||||
" (70, 90, \"#65CBF3\"),\n",
|
||||
" (90, 100, \"#0053D6\"),\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"# --- Find the closest source ---\n",
|
||||
"test_url_pattern = (\n",
|
||||
" \"https://storage.googleapis.com/alphafold-colab{:s}/latest/uniref90_2021_03.fasta.1\"\n",
|
||||
")\n",
|
||||
"ex = futures.ThreadPoolExecutor(3)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def fetch(source):\n",
|
||||
" request.urlretrieve(test_url_pattern.format(source))\n",
|
||||
" return source\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"fs = [ex.submit(fetch, source) for source in [\"\", \"-europe\", \"-asia\"]]\n",
|
||||
"source = None\n",
|
||||
"for f in futures.as_completed(fs):\n",
|
||||
" source = f.result()\n",
|
||||
" ex.shutdown()\n",
|
||||
" break\n",
|
||||
"\n",
|
||||
"JACKHMMER_BINARY_PATH = \"/usr/bin/jackhmmer\"\n",
|
||||
"DB_ROOT_PATH = f\"https://storage.googleapis.com/alphafold-colab{source}/latest/\"\n",
|
||||
"# The z_value is the number of sequences in a database.\n",
|
||||
"MSA_DATABASES = [\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"uniref90\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}uniref90_2021_03.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 59,\n",
|
||||
" \"z_value\": 135_301_051,\n",
|
||||
" },\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"smallbfd\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}bfd-first_non_consensus_sequences.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 17,\n",
|
||||
" \"z_value\": 65_984_053,\n",
|
||||
" },\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"mgnify\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}mgy_clusters_2019_05.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 71,\n",
|
||||
" \"z_value\": 304_820_129,\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"# Search UniProt and construct the all_seq features only for heteromers, not homomers.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MULTIMER and len(set(sequences)) > 1:\n",
|
||||
" MSA_DATABASES.extend(\n",
|
||||
" [\n",
|
||||
" # Swiss-Prot and TrEMBL are concatenated together as UniProt.\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"uniprot\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}uniprot_2021_03.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 98,\n",
|
||||
" \"z_value\": 219_174_961 + 565_254,\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"TOTAL_JACKHMMER_CHUNKS = sum(cfg[\"num_streamed_chunks\"] for cfg in MSA_DATABASES)\n",
|
||||
"\n",
|
||||
"MAX_HITS = {\n",
|
||||
" \"uniref90\": 10_000,\n",
|
||||
" \"smallbfd\": 5_000,\n",
|
||||
" \"mgnify\": 501,\n",
|
||||
" \"uniprot\": 50_000,\n",
|
||||
"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_msa(fasta_path):\n",
|
||||
" \"\"\"Searches for MSA for the given sequence using chunked Jackhmmer search.\"\"\"\n",
|
||||
"\n",
|
||||
" # Run the search against chunks of genetic databases.\n",
|
||||
" raw_msa_results = collections.defaultdict(list)\n",
|
||||
" with tqdm.notebook.tqdm(\n",
|
||||
" total=TOTAL_JACKHMMER_CHUNKS, bar_format=TQDM_BAR_FORMAT\n",
|
||||
" ) as pbar:\n",
|
||||
"\n",
|
||||
" def jackhmmer_chunk_callback(i):\n",
|
||||
" pbar.update(n=1)\n",
|
||||
"\n",
|
||||
" for db_config in MSA_DATABASES:\n",
|
||||
" db_name = db_config[\"db_name\"]\n",
|
||||
" pbar.set_description(f\"Searching {db_name}\")\n",
|
||||
" jackhmmer_runner = jackhmmer.Jackhmmer(\n",
|
||||
" binary_path=JACKHMMER_BINARY_PATH,\n",
|
||||
" database_path=db_config[\"db_path\"],\n",
|
||||
" get_tblout=True,\n",
|
||||
" num_streamed_chunks=db_config[\"num_streamed_chunks\"],\n",
|
||||
" streaming_callback=jackhmmer_chunk_callback,\n",
|
||||
" z_value=db_config[\"z_value\"],\n",
|
||||
" )\n",
|
||||
" # Group the results by database name.\n",
|
||||
" raw_msa_results[db_name].extend(jackhmmer_runner.query(fasta_path))\n",
|
||||
"\n",
|
||||
" return raw_msa_results\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"features_for_chain = {}\n",
|
||||
"raw_msa_results_for_sequence = {}\n",
|
||||
"for sequence_index, sequence in enumerate(sequences, start=1):\n",
|
||||
" print(f\"\\nGetting MSA for sequence {sequence_index}\")\n",
|
||||
"\n",
|
||||
" fasta_path = f\"target_{sequence_index}.fasta\"\n",
|
||||
" with open(fasta_path, \"wt\") as f:\n",
|
||||
" f.write(f\">query\\n{sequence}\")\n",
|
||||
"\n",
|
||||
" # Don't do redundant work for multiple copies of the same chain in the multimer.\n",
|
||||
" if sequence not in raw_msa_results_for_sequence:\n",
|
||||
" raw_msa_results = get_msa(fasta_path=fasta_path)\n",
|
||||
" raw_msa_results_for_sequence[sequence] = raw_msa_results\n",
|
||||
" else:\n",
|
||||
" raw_msa_results = copy.deepcopy(raw_msa_results_for_sequence[sequence])\n",
|
||||
"\n",
|
||||
" # Extract the MSAs from the Stockholm files.\n",
|
||||
" # NB: deduplication happens later in pipeline.make_msa_features.\n",
|
||||
" single_chain_msas = []\n",
|
||||
" uniprot_msa = None\n",
|
||||
" for db_name, db_results in raw_msa_results.items():\n",
|
||||
" merged_msa = notebook_utils.merge_chunked_msa(\n",
|
||||
" results=db_results, max_hits=MAX_HITS.get(db_name)\n",
|
||||
" )\n",
|
||||
" if merged_msa.sequences and db_name != \"uniprot\":\n",
|
||||
" single_chain_msas.append(merged_msa)\n",
|
||||
" msa_size = len(set(merged_msa.sequences))\n",
|
||||
" print(\n",
|
||||
" f\"{msa_size} unique sequences found in {db_name} for sequence {sequence_index}\"\n",
|
||||
" )\n",
|
||||
" elif merged_msa.sequences and db_name == \"uniprot\":\n",
|
||||
" uniprot_msa = merged_msa\n",
|
||||
"\n",
|
||||
" notebook_utils.show_msa_info(\n",
|
||||
" single_chain_msas=single_chain_msas, sequence_index=sequence_index\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Turn the raw data into model features.\n",
|
||||
" feature_dict = {}\n",
|
||||
" feature_dict.update(\n",
|
||||
" pipeline.make_sequence_features(\n",
|
||||
" sequence=sequence, description=\"query\", num_res=len(sequence)\n",
|
||||
" )\n",
|
||||
" )\n",
|
||||
" feature_dict.update(pipeline.make_msa_features(msas=single_chain_msas))\n",
|
||||
" # We don't use templates in AlphaFold notebook, add only empty placeholder features.\n",
|
||||
" feature_dict.update(\n",
|
||||
" notebook_utils.empty_placeholder_template_features(\n",
|
||||
" num_templates=0, num_res=len(sequence)\n",
|
||||
" )\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Construct the all_seq features only for heteromers, not homomers.\n",
|
||||
" if (\n",
|
||||
" model_type_to_use == notebook_utils.ModelType.MULTIMER\n",
|
||||
" and len(set(sequences)) > 1\n",
|
||||
" ):\n",
|
||||
" valid_feats = msa_pairing.MSA_FEATURES + (\n",
|
||||
" \"msa_uniprot_accession_identifiers\",\n",
|
||||
" \"msa_species_identifiers\",\n",
|
||||
" )\n",
|
||||
" all_seq_features = {\n",
|
||||
" f\"{k}_all_seq\": v\n",
|
||||
" for k, v in pipeline.make_msa_features([uniprot_msa]).items()\n",
|
||||
" if k in valid_feats\n",
|
||||
" }\n",
|
||||
" feature_dict.update(all_seq_features)\n",
|
||||
"\n",
|
||||
" features_for_chain[protein.PDB_CHAIN_IDS[sequence_index - 1]] = feature_dict\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Do further feature post-processing depending on the model type.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" np_example = features_for_chain[protein.PDB_CHAIN_IDS[0]]\n",
|
||||
"\n",
|
||||
"elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" all_chain_features = {}\n",
|
||||
" for chain_id, chain_features in features_for_chain.items():\n",
|
||||
" all_chain_features[chain_id] = pipeline_multimer.convert_monomer_features(\n",
|
||||
" chain_features, chain_id\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" all_chain_features = pipeline_multimer.add_assembly_features(all_chain_features)\n",
|
||||
"\n",
|
||||
" np_example = feature_processing.pair_and_merge(\n",
|
||||
" all_chain_features=all_chain_features, is_prokaryote=is_prokaryote\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Pad MSA to avoid zero-sized extra_msa.\n",
|
||||
" np_example = pipeline_multimer.pad_msa(np_example, min_num_seq=512)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "9640643486bd"
|
||||
},
|
||||
"source": [
|
||||
"## Run AlphaFold\n",
|
||||
"\n",
|
||||
"Once this cell has been executed, a zip-archive \"prediction.zip\" with the obtained prediction will be saved on the VM, and available for download to your computer in the sidebar. In case you are having issues with the relaxation stage, you can disable it below. Warning: This means that the prediction might have distracting small stereochemical violations."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "XUo6foMQxwS2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"run_relax = True\n",
|
||||
"\n",
|
||||
"# --- Run the model ---\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" model_names = config.MODEL_PRESETS[\"monomer\"] + (\"model_2_ptm\",)\n",
|
||||
"elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" model_names = config.MODEL_PRESETS[\"multimer\"]\n",
|
||||
"\n",
|
||||
"output_dir = \"prediction\"\n",
|
||||
"os.makedirs(output_dir, exist_ok=True)\n",
|
||||
"\n",
|
||||
"plddts = {}\n",
|
||||
"ranking_confidences = {}\n",
|
||||
"pae_outputs = {}\n",
|
||||
"unrelaxed_proteins = {}\n",
|
||||
"\n",
|
||||
"with tqdm.notebook.tqdm(total=len(model_names) + 1, bar_format=TQDM_BAR_FORMAT) as pbar:\n",
|
||||
" for model_name in model_names:\n",
|
||||
" pbar.set_description(f\"Running {model_name}\")\n",
|
||||
"\n",
|
||||
" cfg = config.model_config(model_name)\n",
|
||||
" if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" cfg.data.eval.num_ensemble = 1\n",
|
||||
" elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" cfg.model.num_ensemble_eval = 1\n",
|
||||
" params = data.get_model_haiku_params(model_name, \"./alphafold/data\")\n",
|
||||
" model_runner = model.RunModel(cfg, params)\n",
|
||||
" processed_feature_dict = model_runner.process_features(\n",
|
||||
" np_example, random_seed=0\n",
|
||||
" )\n",
|
||||
" prediction = model_runner.predict(\n",
|
||||
" processed_feature_dict, random_seed=random.randrange(sys.maxsize)\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" mean_plddt = prediction[\"plddt\"].mean()\n",
|
||||
"\n",
|
||||
" if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" if \"predicted_aligned_error\" in prediction:\n",
|
||||
" pae_outputs[model_name] = (\n",
|
||||
" prediction[\"predicted_aligned_error\"],\n",
|
||||
" prediction[\"max_predicted_aligned_error\"],\n",
|
||||
" )\n",
|
||||
" else:\n",
|
||||
" # Monomer models are sorted by mean pLDDT. Do not put monomer pTM models here as they\n",
|
||||
" # should never get selected.\n",
|
||||
" ranking_confidences[model_name] = prediction[\"ranking_confidence\"]\n",
|
||||
" plddts[model_name] = prediction[\"plddt\"]\n",
|
||||
" elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" # Multimer models are sorted by pTM+ipTM.\n",
|
||||
" ranking_confidences[model_name] = prediction[\"ranking_confidence\"]\n",
|
||||
" plddts[model_name] = prediction[\"plddt\"]\n",
|
||||
" pae_outputs[model_name] = (\n",
|
||||
" prediction[\"predicted_aligned_error\"],\n",
|
||||
" prediction[\"max_predicted_aligned_error\"],\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Set the b-factors to the per-residue plddt.\n",
|
||||
" final_atom_mask = prediction[\"structure_module\"][\"final_atom_mask\"]\n",
|
||||
" b_factors = prediction[\"plddt\"][:, None] * final_atom_mask\n",
|
||||
" unrelaxed_protein = protein.from_prediction(\n",
|
||||
" processed_feature_dict,\n",
|
||||
" prediction,\n",
|
||||
" b_factors=b_factors,\n",
|
||||
" remove_leading_feature_dimension=(\n",
|
||||
" model_type_to_use == notebook_utils.ModelType.MONOMER\n",
|
||||
" ),\n",
|
||||
" )\n",
|
||||
" unrelaxed_proteins[model_name] = unrelaxed_protein\n",
|
||||
"\n",
|
||||
" # Delete unused outputs to save memory.\n",
|
||||
" del model_runner\n",
|
||||
" del params\n",
|
||||
" del prediction\n",
|
||||
" pbar.update(n=1)\n",
|
||||
"\n",
|
||||
" # --- AMBER relax the best model ---\n",
|
||||
"\n",
|
||||
" # Find the best model according to the mean pLDDT.\n",
|
||||
" best_model_name = max(\n",
|
||||
" ranking_confidences.keys(), key=lambda x: ranking_confidences[x]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" if run_relax:\n",
|
||||
" pbar.set_description(\"AMBER relaxation\")\n",
|
||||
" amber_relaxer = relax.AmberRelaxation(\n",
|
||||
" max_iterations=0,\n",
|
||||
" tolerance=2.39,\n",
|
||||
" stiffness=10.0,\n",
|
||||
" exclude_residues=[],\n",
|
||||
" max_outer_iterations=3,\n",
|
||||
" )\n",
|
||||
" relaxed_pdb, _, _ = amber_relaxer.process(\n",
|
||||
" prot=unrelaxed_proteins[best_model_name]\n",
|
||||
" )\n",
|
||||
" else:\n",
|
||||
" print(\"Warning: Running without the relaxation stage.\")\n",
|
||||
" relaxed_pdb = protein.to_pdb(unrelaxed_proteins[best_model_name])\n",
|
||||
" pbar.update(n=1) # Finished AMBER relax.\n",
|
||||
"\n",
|
||||
"# Construct multiclass b-factors to indicate confidence bands\n",
|
||||
"# 0=very low, 1=low, 2=confident, 3=very high\n",
|
||||
"banded_b_factors = []\n",
|
||||
"for plddt in plddts[best_model_name]:\n",
|
||||
" for idx, (min_val, max_val, _) in enumerate(PLDDT_BANDS):\n",
|
||||
" if plddt >= min_val and plddt <= max_val:\n",
|
||||
" banded_b_factors.append(idx)\n",
|
||||
" break\n",
|
||||
"banded_b_factors = np.array(banded_b_factors)[:, None] * final_atom_mask\n",
|
||||
"to_visualize_pdb = utils.overwrite_b_factors(relaxed_pdb, banded_b_factors)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Write out the prediction\n",
|
||||
"pred_output_path = os.path.join(output_dir, \"selected_prediction.pdb\")\n",
|
||||
"with open(pred_output_path, \"w\") as f:\n",
|
||||
" f.write(relaxed_pdb)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# --- Visualise the prediction & confidence ---\n",
|
||||
"show_sidechains = True\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def plot_plddt_legend():\n",
|
||||
" \"\"\"Plots the legend for pLDDT.\"\"\"\n",
|
||||
" thresh = [\n",
|
||||
" \"Very low (pLDDT < 50)\",\n",
|
||||
" \"Low (70 > pLDDT > 50)\",\n",
|
||||
" \"Confident (90 > pLDDT > 70)\",\n",
|
||||
" \"Very high (pLDDT > 90)\",\n",
|
||||
" ]\n",
|
||||
"\n",
|
||||
" colors = [x[2] for x in PLDDT_BANDS]\n",
|
||||
"\n",
|
||||
" plt.figure(figsize=(2, 2))\n",
|
||||
" for c in colors:\n",
|
||||
" plt.bar(0, 0, color=c)\n",
|
||||
" plt.legend(thresh, frameon=False, loc=\"center\", fontsize=20)\n",
|
||||
" plt.xticks([])\n",
|
||||
" plt.yticks([])\n",
|
||||
" ax = plt.gca()\n",
|
||||
" ax.spines[\"right\"].set_visible(False)\n",
|
||||
" ax.spines[\"top\"].set_visible(False)\n",
|
||||
" ax.spines[\"left\"].set_visible(False)\n",
|
||||
" ax.spines[\"bottom\"].set_visible(False)\n",
|
||||
" plt.title(\"Model Confidence\", fontsize=20, pad=20)\n",
|
||||
" return plt\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Show the structure coloured by chain if the multimer model has been used.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" multichain_view = py3Dmol.view(width=800, height=600)\n",
|
||||
" multichain_view.addModelsAsFrames(to_visualize_pdb)\n",
|
||||
" multichain_style = {\"cartoon\": {\"colorscheme\": \"chain\"}}\n",
|
||||
" multichain_view.setStyle({\"model\": -1}, multichain_style)\n",
|
||||
" multichain_view.zoomTo()\n",
|
||||
" multichain_view.show()\n",
|
||||
"\n",
|
||||
"# Color the structure by per-residue pLDDT\n",
|
||||
"color_map = {i: bands[2] for i, bands in enumerate(PLDDT_BANDS)}\n",
|
||||
"view = py3Dmol.view(width=800, height=600)\n",
|
||||
"view.addModelsAsFrames(to_visualize_pdb)\n",
|
||||
"style = {\"cartoon\": {\"colorscheme\": {\"prop\": \"b\", \"map\": color_map}}}\n",
|
||||
"if show_sidechains:\n",
|
||||
" style[\"stick\"] = {}\n",
|
||||
"view.setStyle({\"model\": -1}, style)\n",
|
||||
"view.zoomTo()\n",
|
||||
"\n",
|
||||
"grid = GridspecLayout(1, 2)\n",
|
||||
"out = Output()\n",
|
||||
"with out:\n",
|
||||
" view.show()\n",
|
||||
"grid[0, 0] = out\n",
|
||||
"\n",
|
||||
"out = Output()\n",
|
||||
"with out:\n",
|
||||
" plot_plddt_legend().show()\n",
|
||||
"grid[0, 1] = out\n",
|
||||
"\n",
|
||||
"display.display(grid)\n",
|
||||
"\n",
|
||||
"# Display pLDDT and predicted aligned error (if output by the model).\n",
|
||||
"if pae_outputs:\n",
|
||||
" num_plots = 2\n",
|
||||
"else:\n",
|
||||
" num_plots = 1\n",
|
||||
"\n",
|
||||
"plt.figure(figsize=[8 * num_plots, 6])\n",
|
||||
"plt.subplot(1, num_plots, 1)\n",
|
||||
"plt.plot(plddts[best_model_name])\n",
|
||||
"plt.title(\"Predicted LDDT\")\n",
|
||||
"plt.xlabel(\"Residue\")\n",
|
||||
"plt.ylabel(\"pLDDT\")\n",
|
||||
"\n",
|
||||
"if num_plots == 2:\n",
|
||||
" plt.subplot(1, 2, 2)\n",
|
||||
" pae, max_pae = list(pae_outputs.values())[0]\n",
|
||||
" plt.imshow(pae, vmin=0.0, vmax=max_pae, cmap=\"Greens_r\")\n",
|
||||
" plt.colorbar(fraction=0.046, pad=0.04)\n",
|
||||
"\n",
|
||||
" # Display lines at chain boundaries.\n",
|
||||
" best_unrelaxed_prot = unrelaxed_proteins[best_model_name]\n",
|
||||
" total_num_res = best_unrelaxed_prot.residue_index.shape[-1]\n",
|
||||
" chain_ids = best_unrelaxed_prot.chain_index\n",
|
||||
" for chain_boundary in np.nonzero(chain_ids[:-1] - chain_ids[1:]):\n",
|
||||
" if chain_boundary.size:\n",
|
||||
" plt.plot([0, total_num_res], [chain_boundary, chain_boundary], color=\"red\")\n",
|
||||
" plt.plot([chain_boundary, chain_boundary], [0, total_num_res], color=\"red\")\n",
|
||||
"\n",
|
||||
" plt.title(\"Predicted Aligned Error\")\n",
|
||||
" plt.xlabel(\"Scored residue\")\n",
|
||||
" plt.ylabel(\"Aligned residue\")\n",
|
||||
"\n",
|
||||
"# Save the predicted aligned error (if it exists).\n",
|
||||
"pae_output_path = os.path.join(output_dir, \"predicted_aligned_error.json\")\n",
|
||||
"if pae_outputs:\n",
|
||||
" # Save predicted aligned error in the same format as the AF EMBL DB.\n",
|
||||
" pae_data = notebook_utils.get_pae_json(pae=pae, max_pae=max_pae.item())\n",
|
||||
" with open(pae_output_path, \"w\") as f:\n",
|
||||
" f.write(pae_data)\n",
|
||||
"\n",
|
||||
"!zip -q -r {output_dir}.zip {output_dir}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "lUQAn5LYC5n4"
|
||||
},
|
||||
"source": [
|
||||
"### Interpreting the prediction\n",
|
||||
"\n",
|
||||
"In general predicted LDDT (pLDDT) is best used for intra-domain confidence, whereas Predicted Aligned Error (PAE) is best used for determining between domain or between chain confidence.\n",
|
||||
"\n",
|
||||
"Please see the [AlphaFold methods paper](https://www.nature.com/articles/s41586-021-03819-2), the [AlphaFold predictions of the human proteome paper](https://www.nature.com/articles/s41586-021-03828-1), and the [AlphaFold-Multimer paper](https://www.biorxiv.org/content/10.1101/2021.10.04.463034v1) as well as [our FAQ](https://alphafold.ebi.ac.uk/faq) on how to interpret AlphaFold predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "jeb2z8DIA4om"
|
||||
},
|
||||
"source": [
|
||||
"## FAQ & Troubleshooting\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"* How do I get a predicted protein structure for my protein?\n",
|
||||
" * Connect the notebook to the Jupyter kernel \"Python 3 (ipykernel)\".\n",
|
||||
" * Paste the amino acid sequence of your protein (without any headers) into the variable sequence_1 in \"Making a Prediction\".\n",
|
||||
" * Run all cells in the notebook, either by running them individually or via \"Kernel\"/\"Restart Kernel and Run All Cells...\"\n",
|
||||
" * The predicted protein structure will be downloaded once all cells have been executed. Note: This can take minutes to hours - see below.\n",
|
||||
"* How long will this take?\n",
|
||||
" * The search against genetic databases can take minutes to hours.\n",
|
||||
" * Running AlphaFold and generating the prediction can take minutes to hours, depending on the length of your protein and on which GPU-type your VM has access to.\n",
|
||||
"* My notebook no longer seems to be doing anything, what should I do?\n",
|
||||
" * Some steps may take minutes to hours to complete.\n",
|
||||
" * If nothing happens or if you receive an error message, try restarting your notebook runtime via \"Kernel\"/\"Restart Kernel and Run All Cells...\".\n",
|
||||
" * If this doesn’t help, try resetting restarting your VM inside the GCloud Console (\"Compute Engine\"/\"VM Instances\").\n",
|
||||
"* How does this compare to the open-source version of AlphaFold?\n",
|
||||
" * This notebook version of AlphaFold searches a selected portion of the BFD dataset and currently doesn’t use templates, so its accuracy is reduced in comparison to the full version of AlphaFold that is described in the [AlphaFold paper](https://doi.org/10.1038/s41586-021-03819-2) and [Github repo](https://github.com/deepmind/alphafold/) (the full version is available via the inference script).\n",
|
||||
"* I received a warning “Notebook requires high RAM”, what do I do?\n",
|
||||
" * In the \"Compute Engine\"/\"VM Instances\" Console menu, you can reconfigure the host VM settings. See [Changing the machine type of a VM instance](https://cloud.google.com/compute/docs/instances/changing-machine-type-of-stopped-instance) for instructions.\n",
|
||||
"* Does this tool install anything on my computer?\n",
|
||||
" * No, everything happens in the VM instance within your Google Cloud project.\n",
|
||||
"* How should I share feedback and bug reports?\n",
|
||||
" * Please share any feedback and bug reports as an [issue](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues) on Github.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Related work\n",
|
||||
"\n",
|
||||
"Take a look at these Colab notebooks provided by the community (please note that these notebooks may vary from our validated AlphaFold system and we cannot guarantee their accuracy):\n",
|
||||
"\n",
|
||||
"* The [ColabFold AlphaFold2 notebook](https://colab.research.google.com/github/sokrypton/ColabFold/blob/main/AlphaFold2.ipynb) by Sergey Ovchinnikov, Milot Mirdita and Martin Steinegger, which uses an API hosted at the Södinglab based on the MMseqs2 server ([Mirdita et al. 2019, Bioinformatics](https://academic.oup.com/bioinformatics/article/35/16/2856/5280135)) for the multiple sequence alignment creation.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "YfPhvYgKC81B"
|
||||
},
|
||||
"source": [
|
||||
"# License and Disclaimer\n",
|
||||
"\n",
|
||||
"This is not an officially-supported Google product.\n",
|
||||
"\n",
|
||||
"This notebook and other information provided is for theoretical modelling only, caution should be exercised in its use. It is provided ‘as-is’ without any warranty of any kind, whether expressed or implied. Information is not intended to be a substitute for professional medical advice, diagnosis, or treatment, and does not constitute medical or other professional advice.\n",
|
||||
"\n",
|
||||
"Copyright 2021 DeepMind Technologies Limited.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## AlphaFold Code License\n",
|
||||
"\n",
|
||||
"Licensed under the Apache License, Version 2.0 (the \"License\"); you may not use this file except in compliance with the License. You may obtain a copy of the License at https://www.apache.org/licenses/LICENSE-2.0.\n",
|
||||
"\n",
|
||||
"Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on an \"AS IS\" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the specific language governing permissions and limitations under the License.\n",
|
||||
"\n",
|
||||
"## Model Parameters License\n",
|
||||
"\n",
|
||||
"The AlphaFold parameters are made available under the terms of the Creative Commons Attribution 4.0 International (CC BY 4.0) license. You can find details at: https://creativecommons.org/licenses/by/4.0/legalcode\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Third-party software\n",
|
||||
"\n",
|
||||
"Use of the third-party software, libraries or code referred to in the [Acknowledgements section](https://github.com/deepmind/alphafold/#acknowledgements) in the AlphaFold README may be governed by separate terms and conditions or license provisions. Your use of the third-party software, libraries or code is subject to any such terms and you should check that you can comply with any applicable restrictions or terms and conditions before use.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Mirrored Databases\n",
|
||||
"\n",
|
||||
"The following databases have been mirrored by DeepMind, and are available with reference to the following:\n",
|
||||
"* UniProt: v2021\\_03 (unmodified), by The UniProt Consortium, available under a [Creative Commons Attribution-NoDerivatives 4.0 International License](http://creativecommons.org/licenses/by-nd/4.0/).\n",
|
||||
"* UniRef90: v2021\\_03 (unmodified), by The UniProt Consortium, available under a [Creative Commons Attribution-NoDerivatives 4.0 International License](http://creativecommons.org/licenses/by-nd/4.0/).\n",
|
||||
"* MGnify: v2019\\_05 (unmodified), by Mitchell AL et al., available free of all copyright restrictions and made fully and freely available for both non-commercial and commercial use under [CC0 1.0 Universal (CC0 1.0) Public Domain Dedication](https://creativecommons.org/publicdomain/zero/1.0/).\n",
|
||||
"* BFD: (modified), by Steinegger M. and Söding J., modified by DeepMind, available under a [Creative Commons Attribution-ShareAlike 4.0 International License](https://creativecommons.org/licenses/by/4.0/). See the Methods section of the [AlphaFold proteome paper](https://www.nature.com/articles/s41586-021-03828-1) for details."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"accelerator": "GPU",
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "AlphaFold.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
# Copyright 2022 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
ARG CUDA_MAJOR=11
|
||||
ARG CUDA_MINOR=0
|
||||
|
||||
FROM gcr.io/deeplearning-platform-release/base-cu110
|
||||
|
||||
ARG CUDA_MAJOR
|
||||
ARG CUDA_MINOR
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
|
||||
build-essential \
|
||||
cmake \
|
||||
cuda-command-line-tools-${CUDA_MAJOR}-${CUDA_MINOR} \
|
||||
git \
|
||||
hmmer \
|
||||
kalign \
|
||||
tzdata \
|
||||
wget \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Compile HHsuite from source.
|
||||
RUN git clone --branch v3.3.0 https://github.com/soedinglab/hh-suite.git /tmp/hh-suite \
|
||||
&& mkdir /tmp/hh-suite/build \
|
||||
&& pushd /tmp/hh-suite/build \
|
||||
&& cmake -DCMAKE_INSTALL_PREFIX=/opt/hhsuite .. \
|
||||
&& make -j 4 && make install \
|
||||
&& ln -s /opt/hhsuite/bin/* /usr/bin \
|
||||
&& popd \
|
||||
&& rm -rf /tmp/hh-suite
|
||||
|
||||
ENV PATH="/opt/conda/bin:$PATH"
|
||||
RUN conda update -qy conda \
|
||||
&& conda install -y -c conda-forge \
|
||||
openmm=7.5.1 \
|
||||
cudatoolkit==${CUDA_VERSION} \
|
||||
pdbfixer \
|
||||
pip \
|
||||
python=3.7
|
||||
|
||||
COPY . /app/alphafold
|
||||
|
||||
# Install pip packages.
|
||||
RUN pip3 install --upgrade pip \
|
||||
&& pip3 install -r /app/alphafold/requirements.txt \
|
||||
&& pip3 install py3Dmol tqdm \
|
||||
&& pip3 install --upgrade jax==0.2.14 jaxlib==0.1.69+cuda${CUDA_MAJOR}${CUDA_MINOR} -f \
|
||||
https://storage.googleapis.com/jax-releases/jax_releases.html
|
||||
|
||||
# Install alphafold.
|
||||
WORKDIR /app/alphafold
|
||||
RUN python setup.py install
|
||||
|
||||
# Apply OpenMM patch.
|
||||
WORKDIR /opt/conda/lib/python3.7/site-packages
|
||||
RUN patch -p0 < /app/alphafold/docker/openmm.patch
|
||||
|
||||
# Creating a tmp location for jackhmmr; not mounting through to host though.
|
||||
RUN sudo mkdir -m 777 --parents /tmp/ramdisk
|
||||
|
||||
# We need to run `ldconfig` first to ensure GPUs are visible, due to some quirk
|
||||
# with Debian. See https://github.com/NVIDIA/nvidia-docker/issues/1399 for
|
||||
# details.
|
||||
# ENTRYPOINT does not support easily running multiple commands, so instead we
|
||||
# write a shell script to wrap them up.
|
||||
WORKDIR /home/jupyter
|
||||
RUN echo '#!/bin/bash\nldconfig\n\'
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
#!/usr/bin/env bash
|
||||
set -e
|
||||
|
||||
# Prod (Publicly viewable)
|
||||
PROJECT=cloud-devrel-public-resources
|
||||
REPOSITORY=alphafold
|
||||
LOCAL_IMAGE=alphafold-on-gcp
|
||||
REMOTE_IMAGE=${LOCAL_IMAGE?}
|
||||
TAG=latest
|
||||
REGISTRY="us-west1-docker.pkg.dev/${PROJECT?}/${REPOSITORY?}/${REMOTE_IMAGE?}:${TAG?}"
|
||||
|
||||
git clone https://github.com/deepmind/alphafold.git
|
||||
|
||||
cp Dockerfile alphafold/docker/Dockerfile
|
||||
cp AlphaFold.ipynb alphafold/notebooks/AlphaFold.ipynb
|
||||
|
||||
cd alphafold && sudo docker build --tag ${LOCAL_IMAGE?}:${TAG?} -f docker/Dockerfile .
|
||||
|
||||
sudo docker tag ${LOCAL_IMAGE?}:${TAG?} ${REGISTRY?}
|
||||
sudo docker push ${REGISTRY?}
|
||||
|
After Width: | Height: | Size: 3.1 KiB |
@@ -0,0 +1,52 @@
|
||||
# Overview
|
||||
*Pluto* is a programming environment for Julia, designed to be interactive and helpful. It provides a familiar notebook interface but it is not a Jupyter notebook. The biggest difference is that Pluto notebooks are reactive, changing a variable or function in one cell causes the cells that depend on that variable or function to be reevaluated. Pluto also provides useful interaction mechanisms that allow users to dynamically interact with the notebooks computation state.
|
||||
|
||||
The JuliaCon 2020 presentation: [Interactive notebooks ~ Pluto.jl]() provides a good introduction to Pluto. The source is at [fonsp/Pluto.jl]()
|
||||
|
||||
# Install Pluto
|
||||
|
||||
## Create a Vertex AI JupyterLab Instance
|
||||
|
||||
1. From the [GCP console](https://console.cloud.google.com) "hamburger menu"
|
||||
|
||||
select Vertex AI > Workbench
|
||||
2. Click NEW NOTEBOOK
|
||||
|
||||
* Choose Python 3 if you won't be using a GPU
|
||||
* Choose Python 3 (CUDA Toolkit xx.y) if you do want use a GPU
|
||||
3. Give the notebook an appropriate name
|
||||
4. Edit Notebook properties if you have special requirements otherwise accept the defaults and click CREATE
|
||||
5. When the notebook instance is ready click OPEN JUPYTERLAB
|
||||
|
||||
## Configure JupyterLab
|
||||
|
||||
1. Open a terminal by clicking the Terminal icon.
|
||||
1. Install the plutoserver
|
||||
pip3 install git+https://github.com/fonsp/pluto-on-jupyterlab.git
|
||||
1. In a browser go to [julialang.org/downloads](https://julialang.org/downloads/)
|
||||
1. In the Current stable release right click on the `Generic Linux on x86 / 64-bit (glibc)` link
|
||||
Select copy link address
|
||||
1. Back in the terminal switch to root via
|
||||
sudo -i
|
||||
1. Download the release to /opt and install julia in /usr/local/bin
|
||||
```bash
|
||||
cd /opt
|
||||
wget <paste the release link address>
|
||||
tar xf <name of the downloaded tar file>
|
||||
ln -s /opt/<julia-x.y.z>/bin/julia /usr/local/bin
|
||||
^d
|
||||
```
|
||||
1. Add the Pluto package to Julia
|
||||
```bash
|
||||
julia
|
||||
julia> ]add Pluto
|
||||
julia> bksp
|
||||
julia> using Pluto
|
||||
julia> ^d
|
||||
```
|
||||
1. From the JupyterLab menu bar select File > Shut Down
|
||||
|
||||
# Start Pluto
|
||||
1. Click OPEN JUPYTERLAB in the Workbench
|
||||
1. In the Notebook section of the Launcher click Pluto.jl
|
||||
1. The welcome to Pluto.jl screen should appear
|
||||
@@ -1,6 +1,6 @@
|
||||
# PyTorch on Google Cloud: Text Classification
|
||||
|
||||
In the PyTorch on Google Cloud series of blog posts, we aim to share how to build, train and deploy PyTorch models at scale and how to create reproducible machine learning pipelines on Google Cloud with [Vertex AI](https://cloud.google.com/vertex-ai).
|
||||
In the PyTorch on Google Cloud series of blog posts, we aim to share how to build, train, deploy and orchestrate PyTorch models at scale and how to create reproducible machine learning pipelines on Google Cloud with [Vertex AI](https://cloud.google.com/vertex-ai).
|
||||
|
||||
This tutorial on text classification shows how to train a PyTorch based text classification model by fine tuning a pre-trained Huggingface Transformers model and deploy the model on [Vertex AI](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) using Vertex SDK and [`gcloud ai`](https://cloud.google.com/sdk/gcloud/reference/beta/ai).
|
||||
|
||||
@@ -9,6 +9,7 @@ This tutorial on text classification shows how to train a PyTorch based text cla
|
||||
| <h4>Notebook</h4> | <h4>Description</h4> |
|
||||
| :-------- | :------- |
|
||||
| [pytorch-text-classification-vertex-ai-train-tune-deploy.ipynb](./pytorch-text-classification-vertex-ai-train-tune-deploy.ipynb) | Notebook to show training, hyper-parameter tuning and deploying a PyTorch model on Vertex AI |
|
||||
| [pytorch-text-classification-vertex-ai-pipelines.ipynb](./pytorch-text-classification-vertex-ai-pipelines.ipynb) | Notebook to show orchestration of PyTorch ML workflows on Vertex AI Pipelines using Kubeflow Pipelines SDK |
|
||||
|
||||
## Folders
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
|
||||
# Use pytorch GPU base image
|
||||
FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7
|
||||
# FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7
|
||||
FROM us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-10:latest
|
||||
|
||||
# set working directory
|
||||
WORKDIR /app
|
||||
|
||||
@@ -22,15 +22,18 @@ PROJECT_ID=$(gcloud config list --format 'value(core.project)')
|
||||
|
||||
# BUCKET_NAME: Change to your bucket name.
|
||||
BUCKET_NAME="[your-bucket-name]" # <-- CHANGE TO YOUR BUCKET NAME
|
||||
BUCKET_NAME=cloud-ai-platform-2f444b6a-a742-444b-b91a-c7519f51bd77
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# JOB_NAME: the name of your job running on AI Platform.
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-cstm-cntr-"
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-cstm-cntr"
|
||||
JOB_NAME=${JOB_PREFIX}-$(date +%Y%m%d%H%M%S)-custom-job
|
||||
|
||||
# This can be a GCS location to a zipped and uploaded package
|
||||
PACKAGE_PATH=./trainer
|
||||
|
||||
# REGION: select a region from https://cloud.google.com/vertex-ai/docs/general/locations#available_regions
|
||||
# or use the default '`us-central1`'. The region is where the job will be run.
|
||||
REGION="us-central1"
|
||||
@@ -41,11 +44,8 @@ JOB_DIR=gs://${BUCKET_NAME}/${JOB_PREFIX}/models/${JOB_NAME}
|
||||
# IMAGE_REPO_NAME: set a local repo name to distinquish our image
|
||||
IMAGE_REPO_NAME=pytorch_gpu_train_finetuned-bert-classifier
|
||||
|
||||
# IMAGE_TAG: an easily identifiable tag for your docker image
|
||||
IMAGE_TAG=latest
|
||||
|
||||
# IMAGE_URI: the complete URI location for Cloud Container Registry
|
||||
CUSTOM_TRAIN_IMAGE_URI=gcr.io/${PROJECT_ID}/${IMAGE_REPO_NAME}:${IMAGE_TAG}
|
||||
CUSTOM_TRAIN_IMAGE_URI=gcr.io/${PROJECT_ID}/${IMAGE_REPO_NAME}
|
||||
|
||||
# Build the docker image
|
||||
docker build --no-cache -f Dockerfile -t $CUSTOM_TRAIN_IMAGE_URI ../python_package
|
||||
@@ -53,11 +53,19 @@ docker build --no-cache -f Dockerfile -t $CUSTOM_TRAIN_IMAGE_URI ../python_packa
|
||||
# Deploy the docker image to Cloud Container Registry
|
||||
docker push ${CUSTOM_TRAIN_IMAGE_URI}
|
||||
|
||||
# worker pool spec
|
||||
worker_pool_spec="\
|
||||
replica-count=1,\
|
||||
machine-type=n1-standard-8,\
|
||||
accelerator-type=NVIDIA_TESLA_V100,\
|
||||
accelerator-count=1,\
|
||||
container-image-uri=${CUSTOM_TRAIN_IMAGE_URI}"
|
||||
|
||||
# Submit Custom Job to Vertex AI
|
||||
gcloud beta ai custom-jobs create \
|
||||
--display-name=${JOB_NAME} \
|
||||
--region ${REGION} \
|
||||
--worker-pool-spec=replica-count=1,machine-type='n1-standard-8',accelerator-type='NVIDIA_TESLA_V100',accelerator-count=1,container-image-uri=${CUSTOM_TRAIN_IMAGE_URI} \
|
||||
--worker-pool-spec="${worker_pool_spec}" \
|
||||
--args="--model-name","finetuned-bert-classifier","--job-dir",$JOB_DIR
|
||||
|
||||
echo "After the job is completed successfully, model files will be saved at $JOB_DIR/"
|
||||
|
||||
|
After Width: | Height: | Size: 45 KiB |
|
After Width: | Height: | Size: 37 KiB |
|
After Width: | Height: | Size: 76 KiB |
|
After Width: | Height: | Size: 74 KiB |
|
After Width: | Height: | Size: 248 KiB |
|
After Width: | Height: | Size: 38 KiB |
|
After Width: | Height: | Size: 123 KiB |
@@ -2,10 +2,13 @@
|
||||
FROM pytorch/torchserve:latest-cpu
|
||||
|
||||
# install dependencies
|
||||
RUN python3 -m pip install --upgrade pip
|
||||
RUN pip3 install transformers
|
||||
|
||||
USER model-server
|
||||
|
||||
# copy model artifacts, custom handler and other dependencies
|
||||
COPY ./custom_text_handler.py /home/model-server/
|
||||
COPY ./custom_handler.py /home/model-server/
|
||||
COPY ./index_to_name.json /home/model-server/
|
||||
COPY ./model/finetuned-bert-classifier/ /home/model-server/
|
||||
|
||||
@@ -21,7 +24,7 @@ EXPOSE 7080
|
||||
EXPOSE 7081
|
||||
|
||||
# create model archive file packaging model artifacts and dependencies
|
||||
RUN torch-model-archiver -f --model-name=finetuned-bert-classifier --version=1.0 --serialized-file=/home/model-server/pytorch_model.bin --handler=/home/model-server/custom_text_handler.py --extra-files "/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json" --export-path=/home/model-server/model-store
|
||||
RUN torch-model-archiver -f --model-name=finetuned-bert-classifier --version=1.0 --serialized-file=/home/model-server/pytorch_model.bin --handler=/home/model-server/custom_handler.py --extra-files "/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json" --export-path=/home/model-server/model-store
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "finetuned-bert-classifier=finetuned-bert-classifier.mar", "--model-store", "/home/model-server/model-store"]
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
|
||||
FROM pytorch/torchserve:latest-cpu
|
||||
|
||||
USER root
|
||||
# run and update some basic packages software packages, including security libs
|
||||
RUN apt-get update && apt-get install -y software-properties-common && add-apt-repository -y ppa:ubuntu-toolchain-r/test && apt-get update && apt-get install -y gcc-9 g++-9 apt-transport-https ca-certificates gnupg curl
|
||||
|
||||
# Install gcloud tools for gsutil as well as debugging
|
||||
RUN echo "deb [signed-by=/usr/share/keyrings/cloud.google.gpg] http://packages.cloud.google.com/apt cloud-sdk main" | tee -a /etc/apt/sources.list.d/google-cloud-sdk.list && curl https://packages.cloud.google.com/apt/doc/apt-key.gpg | apt-key --keyring /usr/share/keyrings/cloud.google.gpg add - && apt-get update -y && apt-get install google-cloud-sdk -y
|
||||
|
||||
USER model-server
|
||||
|
||||
# install dependencies
|
||||
RUN python3 -m pip install --upgrade pip
|
||||
RUN pip3 install transformers
|
||||
|
||||
ARG MODEL_NAME=finetuned-bert-classifier
|
||||
ENV MODEL_NAME="${MODEL_NAME}"
|
||||
|
||||
# health and prediction listener ports
|
||||
ARG AIP_HTTP_PORT=7080
|
||||
ENV AIP_HTTP_PORT="${AIP_HTTP_PORT}"
|
||||
|
||||
ARG MODEL_MGMT_PORT=7081
|
||||
|
||||
# expose health and prediction listener ports from the image
|
||||
EXPOSE "${AIP_HTTP_PORT}"
|
||||
EXPOSE "${MODEL_MGMT_PORT}"
|
||||
EXPOSE 8080 8081 8082 7070 7071
|
||||
|
||||
# create torchserve configuration file
|
||||
USER root
|
||||
RUN echo "service_envelope=json\n" "inference_address=http://0.0.0.0:${AIP_HTTP_PORT}\n" "management_address=http://0.0.0.0:${MODEL_MGMT_PORT}" >> /home/model-server/config.properties
|
||||
USER model-server
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["echo", "AIP_STORAGE_URI=${AIP_STORAGE_URI}", ";", "gsutil", "cp", "-r", "${AIP_STORAGE_URI}/${MODEL_NAME}.mar", "/home/model-server/model-store/", ";", "ls", "-ltr", "/home/model-server/model-store/", ";", "torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "${MODEL_NAME}=${MODEL_NAME}.mar", "--model-store", "/home/model-server/model-store"]
|
||||
@@ -52,7 +52,8 @@ class TransformersClassifierHandler(BaseHandler):
|
||||
with open(mapping_file_path) as f:
|
||||
self.mapping = json.load(f)
|
||||
else:
|
||||
logger.warning('Missing the index_to_name.json file. Inference output will not include class name.')
|
||||
logger.warning('Missing the index_to_name.json file. Inference output will default.')
|
||||
self.mapping = {"0": "Negative", "1": "Positive"}
|
||||
|
||||
self.initialized = True
|
||||
|
||||
@@ -88,4 +89,3 @@ class TransformersClassifierHandler(BaseHandler):
|
||||
|
||||
def postprocess(self, inference_output):
|
||||
return inference_output
|
||||
|
||||
@@ -19,13 +19,19 @@ echo "Submitting Custom Job to Vertex AI to train PyTorch model"
|
||||
|
||||
# BUCKET_NAME: Change to your bucket name
|
||||
BUCKET_NAME="[your-bucket-name]" # <-- CHANGE TO YOUR BUCKET NAME
|
||||
BUCKET_NAME="cloud-ai-platform-2f444b6a-a742-444b-b91a-c7519f51bd77"
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The PyTorch image provided by Vertex AI Training.
|
||||
IMAGE_URI="us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-7:latest"
|
||||
|
||||
# JOB_NAME: the name of your job running on Vertex AI.
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-pkg-ar-"
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-pkg-ar"
|
||||
JOB_NAME=${JOB_PREFIX}-$(date +%Y%m%d%H%M%S)-custom-job
|
||||
|
||||
# REGION: select a region from https://cloud.google.com/vertex-ai/docs/general/locations#available_regions
|
||||
@@ -35,19 +41,21 @@ REGION="us-central1"
|
||||
# JOB_DIR: Where to store prepared package and upload output model.
|
||||
JOB_DIR=gs://${BUCKET_NAME}/${JOB_PREFIX}/model/${JOB_NAME}
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
# worker pool spec
|
||||
worker_pool_spec="\
|
||||
replica-count=1,\
|
||||
machine-type=n1-standard-8,\
|
||||
accelerator-type=NVIDIA_TESLA_V100,\
|
||||
accelerator-count=1,\
|
||||
executor-image-uri=${IMAGE_URI},\
|
||||
python-module=trainer.task,\
|
||||
local-package-path=../python_package/"
|
||||
|
||||
# Submit Custom Job to Vertex AI
|
||||
gcloud beta ai custom-jobs create \
|
||||
--display-name=${JOB_NAME} \
|
||||
--region ${REGION} \
|
||||
--python-package-uris=${PACKAGE_PATH} \
|
||||
--worker-pool-spec=replica-count=1,machine-type='n1-standard-8',accelerator-type='NVIDIA_TESLA_V100',accelerator-count=1,executor-image-uri=${IMAGE_URI},python-module='trainer.task',local-package-path="../python_package/" \
|
||||
--worker-pool-spec="${worker_pool_spec}" \
|
||||
--args="--model-name","finetuned-bert-classifier","--job-dir",$JOB_DIR
|
||||
|
||||
echo "After the job is completed successfully, model files will be saved at $JOB_DIR/"
|
||||
|
||||
@@ -122,6 +122,9 @@ def run(args):
|
||||
# Train / Test the model
|
||||
trainer = train(args, text_classifier, train_dataset, test_dataset)
|
||||
|
||||
metrics = trainer.evaluate(eval_dataset=test_dataset)
|
||||
trainer.save_metrics("all", metrics)
|
||||
|
||||
# Export the trained model
|
||||
trainer.save_model(os.path.join("/tmp", args.model_name))
|
||||
|
||||
|
||||
@@ -63,20 +63,20 @@
|
||||
"- [Training](#Training)\n",
|
||||
" - [Run Training Locally in the Notebook](#Training-locally-in-the-notebook)\n",
|
||||
" - [Run Training Job on Vertex AI](#Training-on-Vertex-AI)\n",
|
||||
" - [Training with pre-built container](#Run-Custom-Job-on-Vertex-Training-with-a-pre-built-container)\n",
|
||||
" - [Training with custom container](#Run-Custom-Job-on-Vertex-Training-with-custom-container)\n",
|
||||
" - [Training with pre-built container](#Run-Custom-Job-on-Vertex-AI-Training-with-a-pre-built-container)\n",
|
||||
" - [Training with custom container](#Run-Custom-Job-on-Vertex-AI-Training-with-custom-container)\n",
|
||||
"- [Tuning](#Hyperparameter-Tuning) \n",
|
||||
" - [Run Hyperparameter Tuning job on Vertex AI](#Run-Hyperparameter-Tuning-Job-on-Vertex-AI)\n",
|
||||
"- [Deploying](#Deploying)\n",
|
||||
" - [Deploying model on Vertex Predictions with custom container](#Deploying-model-on-Vertex-Predictions-with-custom-container)\n",
|
||||
" - [Deploying model on Vertex AI Predictions with custom container](#Deploying-model-on-Vertex AI-Predictions-with-custom-container)\n",
|
||||
"\n",
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud Platform (GCP):\n",
|
||||
"\n",
|
||||
"* [Notebooks](https://cloud.google.com/notebooks)\n",
|
||||
"* [Vertex Training](https://cloud.google.com/vertex-ai/docs/training/custom-training)\n",
|
||||
"* [Vertex Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions)\n",
|
||||
"* [Vertex AI Workbench](https://cloud.google.com/vertex-ai-workbench)\n",
|
||||
"* [Vertex AI Training](https://cloud.google.com/vertex-ai/docs/training/custom-training)\n",
|
||||
"* [Vertex AI Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions)\n",
|
||||
"* [Cloud Storage](https://cloud.google.com/storage)\n",
|
||||
"* [Container Registry](https://cloud.google.com/container-registry)\n",
|
||||
"* [Cloud Build](https://cloud.google.com/build) *[Optional]*\n",
|
||||
@@ -202,9 +202,9 @@
|
||||
"id": "e0c1dcadc2c8"
|
||||
},
|
||||
"source": [
|
||||
"We will be using [Vertex SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) to interact with Vertex AI services. The high-level `aiplatform` library is designed to simplify common data science workflows by using wrapper classes and opinionated defaults. \n",
|
||||
"We will be using [Vertex AI SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) to interact with Vertex AI services. The high-level `aiplatform` library is designed to simplify common data science workflows by using wrapper classes and opinionated defaults. \n",
|
||||
"\n",
|
||||
"#### Install Vertex SDK for Python"
|
||||
"#### Install Vertex AI SDK for Python"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1199,7 +1199,7 @@
|
||||
"source": [
|
||||
"### Run predictions locally with sample examples\n",
|
||||
"\n",
|
||||
"Using the trained model, we can predict the sentiment label for an input text after applying the preprocessing function that was used during the training. We will run the predictions locally in the notebook and later show how you can deploy the model to an endpoint using [TorchServe](https://pytorch.org/serve/) on Vertex Predictions."
|
||||
"Using the trained model, we can predict the sentiment label for an input text after applying the preprocessing function that was used during the training. We will run the predictions locally in the notebook and later show how you can deploy the model to an endpoint using [TorchServe](https://pytorch.org/serve/) on Vertex AI Predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1382,7 +1382,7 @@
|
||||
"id": "f7466d414a0e"
|
||||
},
|
||||
"source": [
|
||||
"### Run Custom Job on Vertex Training with a pre-built container"
|
||||
"### Run Custom Job on Vertex AI Training with a pre-built container"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1395,7 +1395,7 @@
|
||||
"\n",
|
||||
"In this notebook, we are using Hugging Face Datasets and fine tuning a transformer model from Hugging Face Transformers Library for sentiment analysis task using PyTorch. We will use [pre-built container for PyTorch](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers#pytorch) and package the training application code by adding standard Python dependencies - `transformers`, `datasets` and `tqdm` - in the `setup.py` file. \n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1569,7 +1569,7 @@
|
||||
"source": [
|
||||
"#### **Run custom training job on Vertex AI**\n",
|
||||
"\n",
|
||||
"We use [Vertex SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#client_libraries) to create and submit training job to the Vertex training service."
|
||||
"We use [Vertex AI SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#client_libraries) to create and submit training job to the Vertex AI training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1578,7 +1578,7 @@
|
||||
"id": "5d2957ef04fd"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1598,7 +1598,7 @@
|
||||
"id": "6b0fed34b728"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Custom Job to Vertex Training service**"
|
||||
"##### **Configure and submit Custom Job to Vertex AI Training service**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1609,7 +1609,7 @@
|
||||
"source": [
|
||||
"Configure a [Custom Job](https://cloud.google.com/vertex-ai/docs/training/create-custom-job) with the [pre-built container](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers) image for PyTorch and training code packaged as Python source distribution. \n",
|
||||
"\n",
|
||||
"**NOTE:** When using Vertex SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job on Vertex Training service."
|
||||
"**NOTE:** When using Vertex AI SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job on Vertex AI Training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1686,7 +1686,7 @@
|
||||
"\n",
|
||||
"You can monitor the custom job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/training-pipelines/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1798,7 +1798,7 @@
|
||||
"id": "c170d386492b"
|
||||
},
|
||||
"source": [
|
||||
"### Run Custom Job on Vertex Training with custom container"
|
||||
"### Run Custom Job on Vertex AI Training with custom container"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1807,7 +1807,7 @@
|
||||
"id": "035227b6e581"
|
||||
},
|
||||
"source": [
|
||||
"To create a [training job with custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container?hl=hr), you define a `Dockerfile` to install or add the dependencies required for the training job. Then, you build and test your Docker image locally to verify, push the image to Container Registry and submit a Custom Job to Vertex Training service.\n",
|
||||
"To create a [training job with custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container?hl=hr), you define a `Dockerfile` to install or add the dependencies required for the training job. Then, you build and test your Docker image locally to verify, push the image to Container Registry and submit a Custom Job to Vertex AI Training service.\n",
|
||||
"\n",
|
||||
""
|
||||
]
|
||||
@@ -1834,7 +1834,7 @@
|
||||
"%%writefile ./custom_container/Dockerfile\n",
|
||||
"\n",
|
||||
"# Use pytorch GPU base image\n",
|
||||
"FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7\n",
|
||||
"FROM us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-10:latest\n",
|
||||
"\n",
|
||||
"# set working directory\n",
|
||||
"WORKDIR /app\n",
|
||||
@@ -1968,7 +1968,7 @@
|
||||
"id": "a23e5e34bea9"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1988,11 +1988,11 @@
|
||||
"id": "abf1fa4085cb"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Custom Job to Vertex Training service**\n",
|
||||
"##### **Configure and submit Custom Job to Vertex AI Training service**\n",
|
||||
"\n",
|
||||
"Configure a [Custom Job](https://cloud.google.com/vertex-ai/docs/training/create-custom-job) with the [custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container) image with training code and other dependencies\n",
|
||||
"\n",
|
||||
"**NOTE:** When using Vertex SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job to train on Vertex Training."
|
||||
"**NOTE:** When using Vertex AI SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job to train on Vertex AI Training."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2044,7 +2044,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# submit the custom job to Vertex training service\n",
|
||||
"# submit the custom job to Vertex AI training service\n",
|
||||
"model = job.run(\n",
|
||||
" replica_count=1,\n",
|
||||
" machine_type=\"n1-standard-8\",\n",
|
||||
@@ -2065,7 +2065,7 @@
|
||||
"\n",
|
||||
"You can monitor the custom job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/training-pipelines/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2148,11 +2148,11 @@
|
||||
"id": "ba6122f929e3"
|
||||
},
|
||||
"source": [
|
||||
"The training application code for fine-tuning a transformer model for sentiment analysis task uses hyperparameters such as learning rate and weight decay. These hyperparameters control the behavior of the training algorithm and can have a significant effect on the performance of the resulting model. This part of the notebook show how you can automate tuning these hyperparameters with Vertex Training service.\n",
|
||||
"The training application code for fine-tuning a transformer model for sentiment analysis task uses hyperparameters such as learning rate and weight decay. These hyperparameters control the behavior of the training algorithm and can have a significant effect on the performance of the resulting model. This part of the notebook show how you can automate tuning these hyperparameters with Vertex AI Training service.\n",
|
||||
"\n",
|
||||
"We submit a [Hyperparameter Tuning job](https://cloud.google.com/vertex-ai/docs/training/hyperparameter-tuning-overview) to Vertex Training service by packaging the training application code and dependencies in a Docker container and push the container to Google Container Registry, similar to running a Custom Job on Vertex AI with Custom Container.\n",
|
||||
"We submit a [Hyperparameter Tuning job](https://cloud.google.com/vertex-ai/docs/training/hyperparameter-tuning-overview) to Vertex AI Training service by packaging the training application code and dependencies in a Docker container and push the container to Google Container Registry, similar to running a Custom Job on Vertex AI with Custom Container.\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2163,7 +2163,7 @@
|
||||
"source": [
|
||||
"### How hyperparameter tuning works in Vertex AI?\n",
|
||||
"\n",
|
||||
"Following are the high level steps involved in running a Hyperparameter Tuning job on Vertex Training service:\n",
|
||||
"Following are the high level steps involved in running a Hyperparameter Tuning job on Vertex AI Training service:\n",
|
||||
"\n",
|
||||
"- You define the hyperparameters to tune the model along with the metric (or goal) to optimize\n",
|
||||
"- Vertex AI runs multiple trials of your training application with the hyperparameters and limits you specified - maximum number of trials to run and number of parallel trials. \n",
|
||||
@@ -2297,7 +2297,7 @@
|
||||
"source": [
|
||||
"### Run Hyperparameter Tuning Job on Vertex AI\n",
|
||||
"\n",
|
||||
"Before submitting the hyperparameter tuning job to Vertex AI, push the custom container image with training application to Google Cloud Container Registry and then submit the job to Vertex AI. We will be using the same image used for running Custom Job on Vertex Training service."
|
||||
"Before submitting the hyperparameter tuning job to Vertex AI, push the custom container image with training application to Google Cloud Container Registry and then submit the job to Vertex AI. We will be using the same image used for running Custom Job on Vertex AI Training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2326,7 +2326,7 @@
|
||||
"id": "f60fab07d67c"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2346,7 +2346,7 @@
|
||||
"id": "6652aa63ddff"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Hyperparameter Tuning Job to Vertex Training service**\n",
|
||||
"##### **Configure and submit Hyperparameter Tuning Job to Vertex AI Training service**\n",
|
||||
"\n",
|
||||
"Configure a [Hyperparameter Tuning Job](https://cloud.google.com/vertex-ai/docs/training/using-hyperparameter-tuning) with the [custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container) image with training code and other dependencies.\n",
|
||||
"\n",
|
||||
@@ -2374,7 +2374,7 @@
|
||||
"id": "9d46db3a8b23"
|
||||
},
|
||||
"source": [
|
||||
"Define the training arguments with `hp-tune` argument set to `y` so that training application code can report metrics to Vertex"
|
||||
"Define the training arguments with `hp-tune` argument set to `y` so that training application code can report metrics to Vertex AI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2548,7 +2548,7 @@
|
||||
"\n",
|
||||
"You can monitor the hyperparameter tuning job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/hyperparameter-tuning-jobs/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2557,7 +2557,7 @@
|
||||
"id": "ba934b434f03"
|
||||
},
|
||||
"source": [
|
||||
"After the job is finished, you can view and format the results of the hyperparameter tuning Trials (run by Vertex Training service) as a Pandas dataframe"
|
||||
"After the job is finished, you can view and format the results of the hyperparameter tuning Trials (run by Vertex AI Training service) as a Pandas dataframe"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2612,7 +2612,7 @@
|
||||
"id": "5dbccb2b7d32"
|
||||
},
|
||||
"source": [
|
||||
"Now from the results of Trials, you can pick the best performing Trial to deploy to Vertex Predictions"
|
||||
"Now from the results of Trials, you can pick the best performing Trial to deploy to Vertex AI Predictions"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2701,8 +2701,8 @@
|
||||
"JOB_NAME=${JOB_PREFIX}-pytorch-hptune-$(date +%Y%m%d%H%M%S)\n",
|
||||
"echo \"Launching hyperparameter tuning job with display name as \"$JOB_NAME\n",
|
||||
"\n",
|
||||
"# BUCKET_NAME: Change to your bucket name\n",
|
||||
"BUCKET_NAME=$1 # <-- CHANGE TO YOUR BUCKET NAME\n",
|
||||
"# BUCKET_NAME is a required parameter to run the cell.\n",
|
||||
"BUCKET_NAME=$1\n",
|
||||
"\n",
|
||||
"# APP_NAME: get application name\n",
|
||||
"APP_NAME=$2\n",
|
||||
@@ -2711,7 +2711,7 @@
|
||||
"JOB_DIR=${BUCKET_NAME}/${JOB_PREFIX}/model/${JOB_NAME}\n",
|
||||
"\n",
|
||||
"# custom container image URI\n",
|
||||
"CUSTOM_TRAIN_IMAGE_URI=f'gcr.io/'${PROJECT_ID}'/pytorch_gpu_train_'${APP_NAME}\n",
|
||||
"CUSTOM_TRAIN_IMAGE_URI='gcr.io/'${PROJECT_ID}'/pytorch_gpu_train_'${APP_NAME}\n",
|
||||
"\n",
|
||||
"# ========================================================\n",
|
||||
"# create hyperparameter tuning configuration file\n",
|
||||
@@ -2772,20 +2772,20 @@
|
||||
"source": [
|
||||
"## Deploying\n",
|
||||
"\n",
|
||||
"Deploying a PyTorch model on [Vertex Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions) requires to use a custom container that serves online predictions. You will deploy a container running [PyTorch's TorchServe](https://pytorch.org/serve/) tool in order to serve predictions from a fine-tuned transformer model from Hugging Face Transformers for sentiment analysis task. You can then use Vertex Predictions to classify sentiment of input texts. \n",
|
||||
"Deploying a PyTorch model on [Vertex AI Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions) requires to use a custom container that serves online predictions. You will deploy a container running [PyTorch's TorchServe](https://pytorch.org/serve/) tool in order to serve predictions from a fine-tuned transformer model from Hugging Face Transformers for sentiment analysis task. You can then use Vertex AI Predictions to classify sentiment of input texts. \n",
|
||||
"\n",
|
||||
"### Deploying model on Vertex Predictions with custom container\n",
|
||||
"### Deploying model on Vertex AI Predictions with custom container\n",
|
||||
"\n",
|
||||
"To use a custom container to serve predictions from a PyTorch model, you must provide Vertex AI with a Docker container image that runs an HTTP server, such as TorchServe in this case. Please refer to [documentation](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) that describes the container image requirements to be compatible with Vertex Predictions.\n",
|
||||
"To use a custom container to serve predictions from a PyTorch model, you must provide Vertex AI with a Docker container image that runs an HTTP server, such as TorchServe in this case. Please refer to [documentation](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) that describes the container image requirements to be compatible with Vertex AI Predictions.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Essentially, to deploy a PyTorch model on Vertex Predictions following are the steps:\n",
|
||||
"Essentially, to deploy a PyTorch model on Vertex AI Predictions following are the steps:\n",
|
||||
"\n",
|
||||
"1. Package the trained model artifacts including [default](https://pytorch.org/serve/#default-handlers) or [custom](https://pytorch.org/serve/custom_service.html) handlers by creating an archive file using [Torch model archiver](https://github.com/pytorch/serve/tree/master/model-archiver)\n",
|
||||
"2. Build a [custom container](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) compatible with Vertex Predictions to serve the model using Torchserve\n",
|
||||
"3. Upload the model with custom container image to serve predictions as a Vertex Model resource\n",
|
||||
"4. Create a Vertex Endpoint and [deploy the model](https://cloud.google.com/vertex-ai/docs/predictions/deploy-model-api) resource"
|
||||
"2. Build a [custom container](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) compatible with Vertex AI Predictions to serve the model using Torchserve\n",
|
||||
"3. Upload the model with custom container image to serve predictions as a Vertex AI Model resource\n",
|
||||
"4. Create a Vertex AI Endpoint and [deploy the model](https://cloud.google.com/vertex-ai/docs/predictions/deploy-model-api) resource"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2815,7 +2815,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile predictor/custom_text_handler.py\n",
|
||||
"%%writefile predictor/custom_handler.py\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import json\n",
|
||||
@@ -2870,7 +2870,8 @@
|
||||
" with open(mapping_file_path) as f:\n",
|
||||
" self.mapping = json.load(f)\n",
|
||||
" else:\n",
|
||||
" logger.warning('Missing the index_to_name.json file. Inference output will not include class name.')\n",
|
||||
" logger.warning('Missing the index_to_name.json file. Inference output will default.')\n",
|
||||
" self.mapping = {\"0\": \"Negative\", \"1\": \"Positive\"}\n",
|
||||
"\n",
|
||||
" self.initialized = True\n",
|
||||
"\n",
|
||||
@@ -3047,10 +3048,13 @@
|
||||
"FROM pytorch/torchserve:latest-cpu\n",
|
||||
"\n",
|
||||
"# install dependencies\n",
|
||||
"RUN python3 -m pip install --upgrade pip\n",
|
||||
"RUN pip3 install transformers\n",
|
||||
"\n",
|
||||
"USER model-server\n",
|
||||
"\n",
|
||||
"# copy model artifacts, custom handler and other dependencies\n",
|
||||
"COPY ./custom_text_handler.py /home/model-server/\n",
|
||||
"COPY ./custom_handler.py /home/model-server/\n",
|
||||
"COPY ./index_to_name.json /home/model-server/\n",
|
||||
"COPY ./model/$APP_NAME/ /home/model-server/\n",
|
||||
"\n",
|
||||
@@ -3070,7 +3074,7 @@
|
||||
" --model-name=$APP_NAME \\\n",
|
||||
" --version=1.0 \\\n",
|
||||
" --serialized-file=/home/model-server/pytorch_model.bin \\\n",
|
||||
" --handler=/home/model-server/custom_text_handler.py \\\n",
|
||||
" --handler=/home/model-server/custom_handler.py \\\n",
|
||||
" --extra-files \"/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json\" \\\n",
|
||||
" --export-path=/home/model-server/model-store\n",
|
||||
"\n",
|
||||
@@ -3129,7 +3133,7 @@
|
||||
"source": [
|
||||
"#### **Run the container locally** ***[Optional]***\n",
|
||||
"\n",
|
||||
"Before push the container image to Container Registry to use it with Vertex Predictions, you can run it as a container in your local environment to verify that the server works as expected"
|
||||
"Before push the container image to Container Registry to use it with Vertex AI Predictions, you can run it as a container in your local environment to verify that the server works as expected"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3267,9 +3271,9 @@
|
||||
"id": "69477b3a00c0"
|
||||
},
|
||||
"source": [
|
||||
"#### **Deploying the serving container to Vertex Predictions**\n",
|
||||
"#### **Deploying the serving container to Vertex AI Predictions**\n",
|
||||
"\n",
|
||||
"We create a model resource on Vertex AI and deploy the model to a Vertex Endpoints. You must deploy a model to an endpoint before using the model. The deployed model runs the custom container image to serve predictions. "
|
||||
"We create a model resource on Vertex AI and deploy the model to a Vertex AI Endpoints. You must deploy a model to an endpoint before using the model. The deployed model runs the custom container image to serve predictions. "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3300,7 +3304,7 @@
|
||||
"id": "a3da91e19af4"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3437,7 +3441,7 @@
|
||||
"id": "bc4673478269"
|
||||
},
|
||||
"source": [
|
||||
"#### **Invoking the Endpoint with deployed Model using Vertex SDK to make predictions**"
|
||||
"#### **Invoking the Endpoint with deployed Model using Vertex AI SDK to make predictions**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3487,7 +3491,7 @@
|
||||
"source": [
|
||||
"##### **Formatting input for online prediction**\n",
|
||||
"\n",
|
||||
"For online prediction requests, the prediction input instances must be formatted as JSON with base64 encoding as shown here:\n",
|
||||
"This notebook uses [Torchserve's KServe based inference API](https://pytorch.org/serve/inference_api.html#kserve-inference-api) which is also [Vertex AI Predictions compatible format](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements#prediction). For online prediction requests, format the prediction input instances as JSON with base64 encoding as shown here:\n",
|
||||
"\n",
|
||||
"```\n",
|
||||
"[\n",
|
||||
@@ -3560,9 +3564,9 @@
|
||||
},
|
||||
"source": [
|
||||
"##### ***[Optional]*** **Make prediction requests using gcloud CLI**\n",
|
||||
"You can also call the Vertex Endpoint to make predictions using [`gcloud beta ai endpoints predict`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/endpoints/predict). \n",
|
||||
"You can also call the Vertex AI Endpoint to make predictions using [`gcloud beta ai endpoints predict`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/endpoints/predict). \n",
|
||||
"\n",
|
||||
"The following cell shows how to make a prediction request to Vertex Endpoints using `gcloud` CLI: "
|
||||
"The following cell shows how to make a prediction request to Vertex AI Endpoints using `gcloud` CLI: "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3653,12 +3657,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_custom_job = True\n",
|
||||
"delete_hp_tuning_job = True\n",
|
||||
"delete_custom_job = False\n",
|
||||
"delete_hp_tuning_job = False\n",
|
||||
"delete_endpoint = True\n",
|
||||
"delete_model = True\n",
|
||||
"delete_bucket = True\n",
|
||||
"delete_image = True"
|
||||
"delete_model = False\n",
|
||||
"delete_bucket = False\n",
|
||||
"delete_image = False"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3686,7 +3690,7 @@
|
||||
"\n",
|
||||
"client_options = {\"api_endpoint\": API_ENDPOINT}\n",
|
||||
"\n",
|
||||
"# Initialize Vertex SDK\n",
|
||||
"# Initialize Vertex AI SDK\n",
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
]
|
||||
},
|
||||
@@ -3924,7 +3928,7 @@
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" print(f\"Deleting all contents from the bucket {BUCKET_NAME}\")\n",
|
||||
"\n",
|
||||
" shell_output=! gsutil du -as $BUCKET_NAME\n",
|
||||
" shell_output = ! gsutil du -as $BUCKET_NAME\n",
|
||||
" print(\n",
|
||||
" f\"Size of the bucket {BUCKET_NAME} before deleting = {shell_output[0].split()[0]} bytes\"\n",
|
||||
" )\n",
|
||||
@@ -3932,7 +3936,7 @@
|
||||
" # uncomment below line to delete contents of the bucket\n",
|
||||
" # ! gsutil rm -r $BUCKET_NAME\n",
|
||||
"\n",
|
||||
" shell_output=! gsutil du -as $BUCKET_NAME\n",
|
||||
" shell_output = ! gsutil du -as $BUCKET_NAME\n",
|
||||
" if float(shell_output[0].split()[0]) > 0:\n",
|
||||
" print(\n",
|
||||
" \"PLEASE UNCOMMENT LINE TO DELETE BUCKET. CONTENT FROM THE BUCKET NOT DELETED\"\n",
|
||||
|
||||
@@ -188,11 +188,14 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install {USER_FLAG} google-cloud-aiplatform==1.0.1\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-pipeline-components==0.1.3\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-aiplatform\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-pipeline-components\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade kfp\n",
|
||||
"! pip3 install {USER_FLAG} numpy==1.20.3\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tensorflow"
|
||||
"! pip3 install {USER_FLAG} numpy\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tensorflow\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade pillow\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tf-agents\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade fastapi"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -287,7 +290,7 @@
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
@@ -518,6 +521,7 @@
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"from kfp.v2 import compiler, dsl\n",
|
||||
"from kfp.v2.google.client import AIPlatformClient"
|
||||
@@ -561,13 +565,34 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H3530hdGGilo"
|
||||
"id": "895ac243c125"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Dataset parameters\n",
|
||||
"RAW_DATA_PATH = \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" # Location of the MovieLens 100K dataset's \"u.data\" file.\n",
|
||||
"\n",
|
||||
"RAW_DATA_PATH = \"gs://[your-bucket-name]/raw_data/u.data\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "62bfb9a820f6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download the sample data into your RAW_DATA_PATH\n",
|
||||
"! gsutil cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $RAW_DATA_PATH"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H3530hdGGilo"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Pipeline parameters\n",
|
||||
"PIPELINE_NAME = \"movielens-pipeline\" # Pipeline display name.\n",
|
||||
"ENABLE_CACHING = False # Whether to enable execution caching for the pipeline.\n",
|
||||
@@ -635,7 +660,7 @@
|
||||
"source": [
|
||||
"#### Run unit tests on the Generator component\n",
|
||||
"\n",
|
||||
"Before running the command, fill in `RAW_DATA_PATH` in [`src/generator/test_generator_component.py`](src/generator/test_generator_component.py)."
|
||||
"Before running the command, you should update the `RAW_DATA_PATH` in [`src/generator/test_generator_component.py`](src/generator/test_generator_component.py)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -713,12 +738,12 @@
|
||||
"TRAINING_ARTIFACTS_DIR = (\n",
|
||||
" f\"{BUCKET_NAME}/artifacts\" # Root directory for training artifacts.\n",
|
||||
")\n",
|
||||
"TRAINING_REPLICA_COUNT = \"1\" # Number of replica to run the custom training job.\n",
|
||||
"TRAINING_REPLICA_COUNT = 1 # Number of replica to run the custom training job.\n",
|
||||
"TRAINING_MACHINE_TYPE = (\n",
|
||||
" \"n1-standard-4\" # Type of machine to run the custom training job.\n",
|
||||
")\n",
|
||||
"TRAINING_ACCELERATOR_TYPE = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Type of accelerators to run the custom training job.\n",
|
||||
"TRAINING_ACCELERATOR_COUNT = \"0\" # Number of accelerators for the custom training job."
|
||||
"TRAINING_ACCELERATOR_COUNT = 0 # Number of accelerators for the custom training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -769,8 +794,12 @@
|
||||
"TRAINED_POLICY_DISPLAY_NAME = (\n",
|
||||
" \"movielens-trained-policy\" # Display name of the uploaded and deployed policy.\n",
|
||||
")\n",
|
||||
"TRAFFIC_SPLIT = {\"0\": 100}\n",
|
||||
"ENDPOINT_DISPLAY_NAME = \"movielens-endpoint\" # Display name of the prediction endpoint.\n",
|
||||
"ENDPOINT_MACHINE_TYPE = \"n1-standard-4\" # Type of machine of the prediction endpoint."
|
||||
"ENDPOINT_MACHINE_TYPE = \"n1-standard-4\" # Type of machine of the prediction endpoint.\n",
|
||||
"ENDPOINT_REPLICA_COUNT = 1 # Number of replicas of the prediction endpoint.\n",
|
||||
"ENDPOINT_ACCELERATOR_TYPE = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Type of accelerators to run the custom training job.\n",
|
||||
"ENDPOINT_ACCELERATOR_COUNT = 0 # Number of accelerators for the custom training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -900,16 +929,17 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components.experimental.custom_job import utils\n",
|
||||
"from kfp.components import load_component_from_url\n",
|
||||
"\n",
|
||||
"generate_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/generator/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/generator/component.yaml\"\n",
|
||||
")\n",
|
||||
"ingest_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
")\n",
|
||||
"train_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -978,7 +1008,7 @@
|
||||
" bigquery_location=bigquery_location,\n",
|
||||
" bigquery_table_id=bigquery_table_id,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" \n",
|
||||
" # Run the Ingester component.\n",
|
||||
" ingest_task = ingest_op(\n",
|
||||
" project_id=project_id,\n",
|
||||
@@ -988,7 +1018,16 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Run the Trainer component and submit custom job to Vertex AI.\n",
|
||||
" train_task = train_op(\n",
|
||||
" # Convert the train_op component into a Vertex AI Custom Job pre-built component\n",
|
||||
" custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" component_spec=train_op,\n",
|
||||
" replica_count=TRAINING_REPLICA_COUNT,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train_task = custom_job_training_op(\n",
|
||||
" training_artifacts_dir=training_artifacts_dir,\n",
|
||||
" tfrecord_file=ingest_task.outputs[\"tfrecord_file\"],\n",
|
||||
" num_epochs=num_epochs,\n",
|
||||
@@ -996,28 +1035,10 @@
|
||||
" num_actions=num_actions,\n",
|
||||
" tikhonov_weight=tikhonov_weight,\n",
|
||||
" agent_alpha=agent_alpha,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" worker_pool_specs = [\n",
|
||||
" {\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": train_task.container.image,\n",
|
||||
" },\n",
|
||||
" \"replicaCount\": TRAINING_REPLICA_COUNT,\n",
|
||||
" \"machineSpec\": {\n",
|
||||
" \"machineType\": TRAINING_MACHINE_TYPE,\n",
|
||||
" \"acceleratorType\": TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" \"acceleratorCount\": TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" train_task.custom_job_spec = {\n",
|
||||
" \"displayName\": train_task.name,\n",
|
||||
" \"jobSpec\": {\n",
|
||||
" \"workerPoolSpecs\": worker_pool_specs,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Run the Deployer components.\n",
|
||||
" # Upload the trained policy as a model.\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
@@ -1034,11 +1055,14 @@
|
||||
" # Deploy the uploaded, trained policy to the created endpoint. (This operation\n",
|
||||
" # has to occur after both model uploading and endpoint creation complete.)\n",
|
||||
" gcc_aip.ModelDeployOp(\n",
|
||||
" project=project_id,\n",
|
||||
" endpoint=endpoint_create_op.outputs[\"endpoint\"],\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" deployed_model_display_name=TRAINED_POLICY_DISPLAY_NAME,\n",
|
||||
" machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" traffic_split=TRAFFIC_SPLIT,\n",
|
||||
" dedicated_resources_machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_accelerator_type=ENDPOINT_ACCELERATOR_TYPE,\n",
|
||||
" dedicated_resources_accelerator_count=ENDPOINT_ACCELERATOR_COUNT,\n",
|
||||
" dedicated_resources_min_replica_count=ENDPOINT_REPLICA_COUNT,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
@@ -1053,12 +1077,11 @@
|
||||
"# Compile the authored pipeline.\n",
|
||||
"compiler.Compiler().compile(pipeline_func=pipeline, package_path=PIPELINE_SPEC_PATH)\n",
|
||||
"\n",
|
||||
"# Createa Vertex AI client.\n",
|
||||
"api_client = AIPlatformClient(project_id=PROJECT_ID, region=REGION)\n",
|
||||
"\n",
|
||||
"# Create a pipeline run job.\n",
|
||||
"response = api_client.create_run_from_job_spec(\n",
|
||||
" job_spec_path=PIPELINE_SPEC_PATH,\n",
|
||||
"job = aiplatform.PipelineJob(\n",
|
||||
" display_name=f\"{PIPELINE_NAME}-startup\",\n",
|
||||
" template_path=PIPELINE_SPEC_PATH,\n",
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" # Pipeline configs\n",
|
||||
" \"project_id\": PROJECT_ID,\n",
|
||||
@@ -1070,7 +1093,9 @@
|
||||
" \"bigquery_table_id\": BIGQUERY_TABLE_ID,\n",
|
||||
" },\n",
|
||||
" enable_caching=ENABLE_CACHING,\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"job.run()"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1111,7 +1136,11 @@
|
||||
"SIMULATOR_SCHEDULE = \"*/5 * * * *\" # Cloud Scheduler cron job schedule for the Simulator. Eg. \"*/5 * * * *\" means every 5 mins.\n",
|
||||
"SIMULATOR_SCHEDULER_MESSAGE = (\n",
|
||||
" \"simulator-message\" # Cloud Scheduler message for the Simulator.\n",
|
||||
")"
|
||||
")\n",
|
||||
"# TF-Agents RL configs\n",
|
||||
"BATCH_SIZE = 8\n",
|
||||
"RANK_K = 20\n",
|
||||
"NUM_ACTIONS = 20"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1221,7 +1250,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoints = ! gcloud beta ai endpoints list \\\n",
|
||||
"endpoints = ! gcloud ai endpoints list \\\n",
|
||||
" --region=$REGION \\\n",
|
||||
" --filter=display_name=$ENDPOINT_DISPLAY_NAME\n",
|
||||
"print(\"\\n\".join(endpoints), \"\\n\")\n",
|
||||
@@ -1424,13 +1453,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from kfp.components import load_component_from_url\n",
|
||||
"\n",
|
||||
"ingest_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
")\n",
|
||||
"train_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -1481,7 +1508,16 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Run the Trainer component and submit custom job to Vertex AI.\n",
|
||||
" train_task = train_op(\n",
|
||||
" # Convert the train_op component into a Vertex AI Custom Job pre-built component\n",
|
||||
" custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" component_spec=train_op,\n",
|
||||
" replica_count=TRAINING_REPLICA_COUNT,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train_task = custom_job_training_op(\n",
|
||||
" training_artifacts_dir=training_artifacts_dir,\n",
|
||||
" tfrecord_file=ingest_task.outputs[\"tfrecord_file\"],\n",
|
||||
" num_epochs=num_epochs,\n",
|
||||
@@ -1489,28 +1525,10 @@
|
||||
" num_actions=num_actions,\n",
|
||||
" tikhonov_weight=tikhonov_weight,\n",
|
||||
" agent_alpha=agent_alpha,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" worker_pool_specs = [\n",
|
||||
" {\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": train_task.container.image,\n",
|
||||
" },\n",
|
||||
" \"replicaCount\": TRAINING_REPLICA_COUNT,\n",
|
||||
" \"machineSpec\": {\n",
|
||||
" \"machineType\": TRAINING_MACHINE_TYPE,\n",
|
||||
" \"acceleratorType\": TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" \"acceleratorCount\": TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" train_task.custom_job_spec = {\n",
|
||||
" \"displayName\": train_task.name,\n",
|
||||
" \"jobSpec\": {\n",
|
||||
" \"workerPoolSpecs\": worker_pool_specs,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Run the Deployer components.\n",
|
||||
" # Upload the trained policy as a model.\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
@@ -1527,11 +1545,13 @@
|
||||
" # Deploy the uploaded, trained policy to the created endpoint. (This operation\n",
|
||||
" # has to occur after both model uploading and endpoint creation complete.)\n",
|
||||
" gcc_aip.ModelDeployOp(\n",
|
||||
" project=project_id,\n",
|
||||
" endpoint=endpoint_create_op.outputs[\"endpoint\"],\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" deployed_model_display_name=TRAINED_POLICY_DISPLAY_NAME,\n",
|
||||
" machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_accelerator_type=ENDPOINT_ACCELERATOR_TYPE,\n",
|
||||
" dedicated_resources_accelerator_count=ENDPOINT_ACCELERATOR_COUNT,\n",
|
||||
" dedicated_resources_min_replica_count=ENDPOINT_REPLICA_COUNT,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -39,14 +39,15 @@ outputs:
|
||||
- {name: bigquery_table_id, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- (PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'google-cloud-bigquery==2.20.0' 'tensorflow==2.5.0' 'tf-agents==0.8.0' || PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||
python3 -m pip install --quiet --no-warn-script-location 'google-cloud-bigquery==2.20.0'
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' --user) && "$0" "$@"
|
||||
'google-cloud-bigquery==2.20.0' 'pillow' 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
|| PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'google-cloud-bigquery==2.20.0' 'pillow' 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
--user) && "$0" "$@"
|
||||
- sh
|
||||
- -ec
|
||||
- |
|
||||
@@ -296,7 +297,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -20,7 +20,7 @@ outputs:
|
||||
- {name: tfrecord_file, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
@@ -187,7 +187,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
google-cloud-bigquery==2.20.0
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
|
||||
@@ -1,2 +1,4 @@
|
||||
google-cloud-pubsub==2.5.0
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
dataclasses==0.6
|
||||
google-cloud-aiplatform==1.8.1
|
||||
tensorflow==2.5.3
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
@@ -27,14 +27,14 @@ outputs:
|
||||
- {name: training_artifacts_dir, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- (PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' || PIP_DISABLE_PIP_VERSION_CHECK=1 python3
|
||||
-m pip install --quiet --no-warn-script-location 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
--user) && "$0" "$@"
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' 'Pillow' || PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||
python3 -m pip install --quiet --no-warn-script-location 'tensorflow==2.5.0'
|
||||
'tf-agents==0.8.0' 'Pillow' --user) && "$0" "$@"
|
||||
- sh
|
||||
- -ec
|
||||
- |
|
||||
@@ -270,7 +270,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -22,13 +22,13 @@ from src.training import task
|
||||
|
||||
|
||||
# Paths and configurations
|
||||
DATA_PATH = "gs://[your-bucket-name]/[your-dataset-dir]/u.data" # FILL IN
|
||||
DATA_PATH = "gs://[your-bucket-name]/artifacts/u.data" # FILL IN
|
||||
ROOT_DIR = "gs://[your-bucket-name]/artifacts" # FILL IN
|
||||
ARTIFACTS_DIR = "gs://[your-bucket-name]/artifacts" # FILL IN
|
||||
PROFILER_DIR = "gs://[your-bucket-name]/profiler" # FILL IN
|
||||
HPTUNING_RESULT_DIR = "[your-hptuning-result-dir]/" # FILL IN
|
||||
HPTUNING_RESULT_PATH = os.path.join(HPTUNING_RESULT_DIR,
|
||||
"[your-file-name].json") # FILL IN
|
||||
"result.json") # FILL IN
|
||||
RAW_BUCKET_NAME = "[your-hptuning-result-bucket-name]" # FILL IN
|
||||
|
||||
# Hyperparameters
|
||||
|
||||
@@ -1 +1 @@
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
@@ -1 +1 @@
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -113,8 +113,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud beta ai custom-jobs local-run \\\n",
|
||||
" --base-image=$BASE_IMAGE_URI \\\n",
|
||||
"! gcloud ai custom-jobs local-run \\\n",
|
||||
" --executor-image-uri=$BASE_IMAGE_URI \\\n",
|
||||
" --script=$SCRIPT_PATH \\\n",
|
||||
" --output-image-uri=$OUTPUT_IMAGE_NAME \\\n",
|
||||
" -- \\\n",
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
The [official](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/official) folder contains notebooks organized by Google Cloud product.
|
||||
|
||||
The [community](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/community) folder contains notebooks that aren't officially supported by Google.
|
||||
|
||||
Contributions to the repo should use the [notebook template](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb) as a starting point.
|
||||
@@ -3,16 +3,20 @@
|
||||
# @global-owner1 and @global-owner2 will be requested for
|
||||
# review when someone opens a pull request.
|
||||
|
||||
/sdk/sdk_* @aferlitsch
|
||||
/gapic @aferlitsch
|
||||
/ml_ops @aferlitsch
|
||||
/model_monitoring/* @mco
|
||||
/sdk/sdk_* @andrewferlitsch
|
||||
/gapic @andrewferlitsch
|
||||
/ml_ops @andrewferlitsch
|
||||
/model_monitoring/* @mco-gh
|
||||
/structured_data/rapid_prototyping_* @rafael-carvalho
|
||||
|
||||
/managed_notebooks/ @notebooks-team
|
||||
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
|
||||
/managed_notebooks/
|
||||
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
|
||||
/pipelines/google_cloud_pipeline_components_TPU_model_train_upload_deploy.ipynb @brianchunkang
|
||||
/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb @thehardikv
|
||||
/sdk/sdk_automl_forecasting_evaluating_a_model.ipynb @thehardikv
|
||||
/matching_engine @yinghsienwu
|
||||
/neo4j @benofben @htappen
|
||||
/matching_engine/sdk_matching_engine_for_indexing.ipynb @ivanmkc
|
||||
/matching_engine/matching_engine_for_indexing.ipynb @yinghsienwu
|
||||
/sdk/pytorch_lightning_custom_container_training.ipynb @brianchunkang
|
||||
/tensorboard @yfang1
|
||||
/feature_store @nayaknishant @morgandu
|
||||
/vertex_endpoints/tf_hub_obj_detection/deploy_tfhub_object_detection_on_vertex_endpoints.ipynb @entrpn
|
||||
/vertex_endpoints/nvidia-triton/nvidia-triton-custom-container-prediction.ipynb @RajeshThallam
|
||||
/vertex_endpoints/optimized_tensorflow_runtime @vlasenkoalexey
|
||||
|
||||
|
After Width: | Height: | Size: 83 KiB |
|
After Width: | Height: | Size: 141 KiB |
|
After Width: | Height: | Size: 230 KiB |
|
After Width: | Height: | Size: 122 KiB |
|
After Width: | Height: | Size: 140 KiB |
@@ -6,16 +6,17 @@
|
||||
"id": "c8c4e360024a"
|
||||
},
|
||||
"source": [
|
||||
"# Taxi fare prediction using chicago taxi-cab dataset\n",
|
||||
"# Taxi fare prediction using the Chicago Taxi Trips dataset\n",
|
||||
"\n",
|
||||
"## Table of contents\n",
|
||||
"\n",
|
||||
"* [Overview](#section-1)\n",
|
||||
"* [Dataset](#section-2)\n",
|
||||
"* [Objective](#section-3)\n",
|
||||
"* [Costs](#section-4)\n",
|
||||
"* [Data analysis](#section-5)\n",
|
||||
"* [Fit a simple linear regression model](#section-6)\n",
|
||||
"* [Save the model and upload to a GCS bucket](#section-7)\n",
|
||||
"* [Save the model and upload to a Cloud Storage bucket](#section-7)\n",
|
||||
"* [Deploy the model on Vertex AI with support for Vertex Explainable AI](#section-8)\n",
|
||||
"* [Get explanations from the deployed model](#section-9)\n",
|
||||
"* [Clean up](#section-10)\n",
|
||||
@@ -23,23 +24,23 @@
|
||||
"## Overview\n",
|
||||
"<a name=\"section-1\"></a>\n",
|
||||
"\n",
|
||||
"This notebooks demonstrates analysis, feature selection, model building and deployment with Vertex Explainable AI configured on Vertex AI on a subset of the Chicago Taxi-cab dataset for Taxi-fare prediction problem.\n",
|
||||
"This notebook demonstrates analysis, feature selection, model building, and deployment with Vertex Explainable AI configured on Vertex AI, using a subset of the Chicago Taxi Trips dataset for taxi-fare prediction.\n",
|
||||
"\n",
|
||||
"Note: This notebook file was developed to run in a [Vertex AI Workbench managed notebooks](https://console.cloud.google.com/vertex-ai/workbench/list/managed) instance using the Python(Local) kernel. Some components of this notebook may not work in other notebook environments.\n",
|
||||
"*Note: This notebook file was developed to run in a [Vertex AI Workbench managed notebooks](https://console.cloud.google.com/vertex-ai/workbench/list/managed) instance using the Python (Local) kernel. Some components of this notebook may not work in other notebook environments.*\n",
|
||||
"\n",
|
||||
"## Dataset\n",
|
||||
"<a name=\"section-2\"></a>\n",
|
||||
"\n",
|
||||
"The Chicago Taxi-cab dataset includes taxi trips from 2013 to the present, reported to the City of Chicago in its role as a regulatory agency. To protect privacy but allow for aggregate analyses, the Taxi ID is consistent for any given taxi medallion number but does not show the number, Census Tracts are suppressed in some cases, and times are rounded to the nearest 15 minutes. Due to the data reporting process, not all trips are reported but the City believes that most are. This dataset is publicly available on Bigquery under the public datasets with the Table ID : `bigquery-public-data.chicago_taxi_trips.taxi_trips` and also as public dataset on Kaggle Datasets at : [Chicago Taxi Trips Dataset](https://www.kaggle.com/chicago/chicago-taxi-trips-bq).\n",
|
||||
"The Chicago Taxi Trips dataset includes taxi trips from 2013 to the present, reported to the city of Chicago in its role as a regulatory agency. To protect privacy but allow for aggregate analyses, the taxi ID is consistent for any given taxi medallion number but does not show the number, census tracts are suppressed in some cases, and times are rounded to the nearest 15 minutes. Due to the data reporting process, not all trips are reported but the city believes that most are. This dataset is publicly available on BigQuery as a public dataset with the table ID `bigquery-public-data.chicago_taxi_trips.taxi_trips` and also as a public dataset on Kaggle at [Chicago Taxi Trips](https://www.kaggle.com/chicago/chicago-taxi-trips-bq).\n",
|
||||
"\n",
|
||||
" For more information about this dataset and how it was created, please refer [Chicago Digital website](http://digital.cityofchicago.org/index.php/chicago-taxi-data-released).\n",
|
||||
"For more information about this dataset and how it was created, see the [Chicago Digital website](http://digital.cityofchicago.org/index.php/chicago-taxi-data-released).\n",
|
||||
"\n",
|
||||
"## Objective\n",
|
||||
"<a name=\"section-3\"></a>\n",
|
||||
"\n",
|
||||
"The goal of this notebook is to provide an overview on the latest Vertex AI features like Explainable AI and Bigquery in Notebook by trying to solve a Taxi-fare prediction problem. The steps followed in this notebook include : \n",
|
||||
"The goal of this notebook is to provide an overview on the latest Vertex AI features like Explainable AI and \"BigQuery in Notebooks\" by trying to solve a taxi fare prediction problem. The steps followed in this notebook include: \n",
|
||||
"\n",
|
||||
"- Loading the dataset using `Bigquery in Notebooks`.\n",
|
||||
"- Loading the dataset using \"BigQuery in Notebooks\".\n",
|
||||
"- Performing exploratory data analysis on the dataset.\n",
|
||||
"- Feature selection and preprocessing.\n",
|
||||
"- Building a linear regression model using scikit-learn.\n",
|
||||
@@ -54,12 +55,12 @@
|
||||
"This tutorial uses the following billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Bigquery\n",
|
||||
"- BigQuery\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing), [Bigquery pricing](https://cloud.google.com/bigquery/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing), [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
@@ -71,7 +72,9 @@
|
||||
"id": "5ed1f5e85640"
|
||||
},
|
||||
"source": [
|
||||
"#### Set your project ID\n",
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
@@ -84,6 +87,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
@@ -120,11 +125,11 @@
|
||||
"id": "fed4b24ea061"
|
||||
},
|
||||
"source": [
|
||||
"## Select or Create Cloud Storage Bucket for storing the model\n",
|
||||
"## Select or create a Cloud Storage bucket for storing the model\n",
|
||||
"\n",
|
||||
"When you create a model resource on Vertex AI using the Cloud SDK, you need to give a Cloud Storage bucket uri of the model where the model is stored. Using the model saved, you can then create Vertex AI model and endpoint resources in order to serve online predictions.\n",
|
||||
"When you create a model resource on Vertex AI using the Cloud SDK, you need to give a Cloud Storage bucket uri of the model where the model is stored. Using the model saved, you can then create a Vertex AI model and endpoint resources in order to serve online predictions.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.You may also change the REGION variable, which is used for operations throughout the rest of this notebook. Make sure to choose a region where Vertex AI services are available."
|
||||
"Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. You may also change the `LOCATION` variable, which is used for operations throughout the rest of this notebook. Make sure to choose a region where Vertex AI services are available."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -148,7 +153,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Set a default bucketname in case bucket name is not given\n",
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"# Set a default bucket name in case bucket name is not given\n",
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME == \"[your-bucket-name]\" or BUCKET_NAME is None:\n",
|
||||
"\n",
|
||||
" TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n",
|
||||
@@ -165,15 +172,6 @@
|
||||
"<b>Only if your bucket doesn't already exist</b>: Run the following cell to create your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "95702536e547"
|
||||
},
|
||||
"source": [
|
||||
"## Import the required libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -205,6 +203,15 @@
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2e52fd6d4854"
|
||||
},
|
||||
"source": [
|
||||
"## Import the required libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -231,12 +238,14 @@
|
||||
"id": "5166f42557ad"
|
||||
},
|
||||
"source": [
|
||||
"The dataset is quite a large and noisy one and so data from a specific date range will be used. Based on various blogs and resources that are available online, many of them seem to have used the data from around May-2018 which gave some really good results compared to the other date ranges. While there are also some complicated research models propsed for the same problem like considering the weather data, holidays and seasons etc., the current notebook only explores a simple linear regression model as our main objective is to demonstrate the model deployment with Vertex Explainable AI configured on Vertex AI.\n",
|
||||
"The dataset is quite a large and noisy one, so data from a specific date range will be used. Based on various blogs and resources that are available online, many of them seem to have used the data from around May 2018 which gave some really good results compared to the other date ranges. While there are also some complicated research models proposed for the same problem, like considering the weather data, holidays and seasons, the current notebook only explores a simple linear regression model, as our main objective is to demonstrate the model deployment with Vertex Explainable AI configured on Vertex AI.\n",
|
||||
"\n",
|
||||
"## Accessing the data through Bigquery in Notebooks\n",
|
||||
"`Bigquery in Notebooks` feature of Vertex AI's managed notebooks allows us to use Bigquery and its features from the notebook itself eliminating the need to switch between tabs everytime. For every cell in the notebook, there is an option for Bigquery integration at the top right selecting which would enable us to compose a SQL query that can be executed in Bigquery. \n",
|
||||
"## Accessing the data through \"BigQuery in Notebooks\"\n",
|
||||
"\n",
|
||||
"The \"BigQuery in Notebooks\" feature of Vertex AI Workbench managed notebooks lets you use BigQuery and its features from the notebook itself eliminating the need to switch between tabs everytime. For every cell in the notebook, there is an option for the BigQuery integration at the top right, and selecting it enables you to compose an SQL query that can be executed in BigQuery. \n",
|
||||
"\n",
|
||||
"The chosen dataset consists of the following fields:\n",
|
||||
"\n",
|
||||
"The chosen dataset consists of the following fields :\n",
|
||||
"- `unique_key` : Unique identifier for the trip.\n",
|
||||
"- `taxi_id` : A unique identifier for the taxi.\n",
|
||||
"- `trip_start_timestamp`: When the trip started, rounded to the nearest 15 minutes.\n",
|
||||
@@ -261,12 +270,12 @@
|
||||
"- `dropoff_longitude`: The longitude of the center of the dropoff census tract or the community area if the census tract has been hidden for privacy.\n",
|
||||
"- `dropoff_location`: The location of the center of the dropoff census tract or the community area if the census tract has been hidden for privacy.\n",
|
||||
"\n",
|
||||
"Among the available fields in the dataset, only the fields that seem common and relevant for analysis and modeling like `taxi_id`, `trip_start_timestamp`, `trip_seconds`, `trip_miles`, `payment_type` and `trip_total` are selected. Further, the field `trip_total` is treated as the target variable that would be predicted by the machine learning model. Apparently, this field is a summation of `fare`,`tips`,`tolls` and `extras` fields and so because of their correlation with the target variable, they are being excluded for modeling. Due to the volume of the data, a subset of the dataset over the course of one week i.e., 12-May-2018 to 18-May-2018 is being considered. Within this date range itself, the datapoints can be noisy and so a few conditions like the following are considered : \n",
|
||||
"Among the available fields in the dataset, only the fields that seem common and relevant for analysis and modeling like `taxi_id`, `trip_start_timestamp`, `trip_seconds`, `trip_miles`, `payment_type` and `trip_total` are selected. Further, the field `trip_total` is treated as the target variable that would be predicted by the machine learning model. Apparently, this field is a summation of the `fare`,`tips`,`tolls` and `extras` fields and so because of their correlation with the target variable, they are being excluded for modeling. Due to the volume of the data, a subset of the dataset over the course of one week, 12-May-2018 to 18-May-2018 is being considered. Within this date range itself, the datapoints can be noisy and so a few conditions like the following are considered: \n",
|
||||
"\n",
|
||||
"- Time taken for the trip > 0.\n",
|
||||
"- Distance covered during the trip > 0.\n",
|
||||
"- Total trip charges > 0 and\n",
|
||||
"- Pickup and dropoff areas are valid(not empty)."
|
||||
"- Pickup and dropoff areas are valid (not empty)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -301,7 +310,7 @@
|
||||
"id": "781341730c28"
|
||||
},
|
||||
"source": [
|
||||
"The Bigquery integration also allows us to load the queried data into a pandas dataframe using the `Query and load as DataFrame` button. Clicking the button adds a new cell below that provides a code snippet to load the data into a dataframe."
|
||||
"The BigQuery integration also lets you load the queried data into a pandas dataframe using the `Query and load as DataFrame` button. Clicking the button adds a new cell below that provides a code snippet to load the data into a dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -343,7 +352,7 @@
|
||||
"id": "96d61011e159"
|
||||
},
|
||||
"source": [
|
||||
"Check the fields in the data and the shape."
|
||||
"Check the fields in the data and their shape."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -426,7 +435,7 @@
|
||||
"id": "f0feadc628e4"
|
||||
},
|
||||
"source": [
|
||||
"Depending on the percentage of null values in the data, one can choose to either drop them or impute them with mean/median(for numerical values) and mode(for categorical values). In the current data, there doesn't seem to be any null values."
|
||||
"Depending on the percentage of null values in the data, one can choose to either drop them or impute them with mean/median (for numerical values) and mode (for categorical values). In the current data, there doesn't seem to be any null values."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -480,7 +489,7 @@
|
||||
"## Analyze numerical data\n",
|
||||
"<a name=\"section-5\"></a>\n",
|
||||
"\n",
|
||||
"To further anaylyze the data, there are various plots that can be used on numerical and categorical fields. In case of numerical data, one can use histograms and box-plots while bar charts are suited for categorical data to better understand the distribution of the data and the outliers in the data."
|
||||
"To further anaylyze the data, there are various plots that can be used on numerical and categorical fields. In case of numerical data, one can use histograms and box plots while bar charts are suited for categorical data to better understand the distribution of the data and the outliers in the data."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -489,7 +498,7 @@
|
||||
"id": "fa2d6258b509"
|
||||
},
|
||||
"source": [
|
||||
"Plot Histograms and Box-plots on the numerical fields."
|
||||
"Plot histograms and box plots on the numerical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -515,7 +524,7 @@
|
||||
"id": "c3672976d67b"
|
||||
},
|
||||
"source": [
|
||||
"The field `trip_seconds` describes the time taken for the trip in seconds. Optionally, it can be converted into hours for an easier understanding."
|
||||
"The field `trip_seconds` describes the time taken for the trip in seconds. Optionally, it can be converted into hours."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -557,7 +566,7 @@
|
||||
"id": "58d57879aa8a"
|
||||
},
|
||||
"source": [
|
||||
"So far we've only considered to look at the univariate plots. To better understand the relationship between the variables, a pair-plot can be plotted."
|
||||
"So far you've only looked at the univariate plots. To better understand the relationship between the variables, a pair-plot can be plotted."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -580,7 +589,7 @@
|
||||
"id": "b69e8094ba39"
|
||||
},
|
||||
"source": [
|
||||
"From the box-plots and the histograms plotted so far, it is evident that there are some outliers causing skewness in the data which perhaps could be removed. Also, we can certainly see some linear relationship between the independent variables considered in the pair-plot i.e., `trip_seconds` and `trip_miles` and the dependant variable `trip_total`."
|
||||
"From the box plots and the histograms visualized so far, it is evident that there are some outliers causing skewness in the data which perhaps could be removed. Also, you can see some linear relationships between the independent variables considered in the pair-plot, for example, `trip_seconds` and `trip_miles` and the dependant variable `trip_total`."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -627,7 +636,7 @@
|
||||
"id": "341b581e2155"
|
||||
},
|
||||
"source": [
|
||||
"## Analyze Categorical data\n",
|
||||
"## Analyze categorical data\n",
|
||||
"\n",
|
||||
"Further, explore the categorical data by plotting the distribution of all the levels in each field."
|
||||
]
|
||||
@@ -653,9 +662,9 @@
|
||||
"id": "a40a4b2d9d6a"
|
||||
},
|
||||
"source": [
|
||||
"From the above analysis, one can see that almost 99% of the transaction types are Cash and Credit Card. While there are also other type of transactions, their distribution is very less. In such a case, the lower distribution levels can be dropped. On the other hand, total number of pickup and dropoff community areas both seem to have the same levels which make sense. In this case also, one can choose to omit the lower distribution levels but it has to be made sure that both the fields have the same levels afterwards. In the current notebook, we'd keep them as is and proceed with the modeling.\n",
|
||||
"From the above analysis, one can see that almost 99% of the transaction types are Cash and Credit Card. While there are also other type of transactions, their distribution is negligible. In such a case, the lower distribution levels can be dropped. On the other hand, the total number of pickup and dropoff community areas both seem to have the same levels which make sense. In this case also, one can choose to omit the lower distribution levels but you'd have to make sure that both the fields have the same levels afterward. In the current notebook, keep them as is and proceed with the modeling.\n",
|
||||
"\n",
|
||||
"The relationships between the target variable and the categorical fields can be represented through boxplots. For each level, the corresponding distribution of the target variable can be identified."
|
||||
"The relationships between the target variable and the categorical fields can be represented through box plots. For each level, the corresponding distribution of the target variable can be identified."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -680,7 +689,7 @@
|
||||
"id": "f49125a8a866"
|
||||
},
|
||||
"source": [
|
||||
"There seems to be one case where the `trip_total` is over 3000 and has the same pickup and dropoff community area i.e., 28 which is clearly an outlier compared to the rest of the points. This datapoint can be removed."
|
||||
"There seems to be one case where the `trip_total` is over 3000 and has the same pickup and dropoff community area: 28 is clearly an outlier compared to the rest of the points. This datapoint can be removed."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -725,7 +734,7 @@
|
||||
"id": "58a1d9f0a122"
|
||||
},
|
||||
"source": [
|
||||
"There are also timestamp fields in the data that can prove to be useful. `trip_start_timestamp` represents the start timestamp of the taxi-trip and fields like what day of week it was and what hour it was can be dervied from it."
|
||||
"There are also useful timestamp fields in the data. `trip_start_timestamp` represents the start timestamp of the taxi trip and fields like what day of week it was and what hour it was can be derived from it."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -747,7 +756,7 @@
|
||||
"id": "30ae02a15aa1"
|
||||
},
|
||||
"source": [
|
||||
"Since the current dataset is considered only for a week, if there isn't much variation in the newly dervied fields with respect to the target variable, they can be dropped.\n",
|
||||
"Since the current dataset is limited to only a week, if there isn't much variation in the newly derived fields with respect to the target variable, they can be dropped.\n",
|
||||
"\n",
|
||||
"Plot sum and average of the `trip_total` with respect to the `dayofweek`."
|
||||
]
|
||||
@@ -804,9 +813,9 @@
|
||||
"id": "739e985af704"
|
||||
},
|
||||
"source": [
|
||||
"As these plots don't seem to have constant figures with respect to the target variable across their levels, they can be considered for training. In fact, to simplify things these dervied features can be bucketed into less number of levels.\n",
|
||||
"As these plots don't seem to have constant figures with respect to the target variable across their levels, they can be considered for training. In fact, to simplify things these derived features can be bucketed into fewer levels.\n",
|
||||
"\n",
|
||||
"`dayofweek` field can be bucketed into a binary field considering whether or not it was a weekend. If it is a weekday, the record can be assigned 1, else 0. Similarly, `hour` field can also be bucketed and encoded. The normal working hours in Chicago can be assumed to be between *8AM*-*10PM* and if the value falls in between the working hours, it can be encoded as 1, else 0."
|
||||
"The `dayofweek` field can be bucketed into a binary field considering whether or not it was a weekend. If it is a weekday, the record can be assigned 1, else 0. Similarly, the `hour` field can also be bucketed and encoded. The normal working hours in Chicago can be assumed to be between *8AM*-*10PM* and if the value falls in between the working hours, it can be encoded as 1, else 0."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -848,7 +857,7 @@
|
||||
"id": "fe87612faa94"
|
||||
},
|
||||
"source": [
|
||||
"## Divide the data in Train and Test sets\n",
|
||||
"## Divide the data into train and test sets\n",
|
||||
"\n",
|
||||
"Split the preprocessed dataset into train and test sets so that the linear regression model can be validated on the test set."
|
||||
]
|
||||
@@ -887,10 +896,10 @@
|
||||
"id": "5b7e470de1da"
|
||||
},
|
||||
"source": [
|
||||
"## Fit a Simple Linear Regression model\n",
|
||||
"## Fit a simple linear regression model\n",
|
||||
"<a name=\"section-6\"></a>\n",
|
||||
"\n",
|
||||
"Fit a linear regression model using Sklearn's LinearRegression method on the train data."
|
||||
"Fit a linear regression model using scikit-learn's LinearRegression method on the train data."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -940,7 +949,7 @@
|
||||
"id": "2ef6b44f0f93"
|
||||
},
|
||||
"source": [
|
||||
"A low RMSE error and a train and test R2 score of 0.93 suggests that the model has fitted well on the data. Further, the coefficients learned by the model for each of its independent variables can also be checked by checking the `coef_` attribute of the sklearn model. \n",
|
||||
"A low RMSE error and a train and test R2 score of 0.93 suggests that the model is fitted well. Further, the coefficients learned by the model for each of its independent variables can also be checked by checking the `coef_` attribute of the sklearn model. \n",
|
||||
"\n",
|
||||
"Check the coefficients learned by the model."
|
||||
]
|
||||
@@ -963,7 +972,7 @@
|
||||
"id": "bcaed0b52e60"
|
||||
},
|
||||
"source": [
|
||||
"## Save the model and upload to a GCS bucket.\n",
|
||||
"## Save the model and upload to a Cloud Storage bucket\n",
|
||||
"<a name=\"section-7\"></a>\n",
|
||||
"\n",
|
||||
"To deploy the model on Vertex AI, the model needs to be stored in a Cloud Storage bucket first."
|
||||
@@ -999,10 +1008,10 @@
|
||||
"id": "9f8ecfa6a19b"
|
||||
},
|
||||
"source": [
|
||||
"## Deploy the Model on Vertex AI with support for Vertex Explainable AI\n",
|
||||
"## Deploy the model on Vertex AI with support for Vertex Explainable AI\n",
|
||||
"<a name=\"section-8\"></a>\n",
|
||||
"\n",
|
||||
"Configure the Vertex Explainable AI before deploying the model. For further details, see [Configuring Vertex Explainable AI in Vertex AI models](https://cloud.google.com/vertex-ai/docs/explainable-ai/configuring-explanations#scikit-learn-and-xgboost-pre-built-containers)."
|
||||
"Configure Vertex Explainable AI before deploying the model. For further details, see [Configuring Vertex Explainable AI in Vertex AI models](https://cloud.google.com/vertex-ai/docs/explainable-ai/configuring-explanations#scikit-learn-and-xgboost-pre-built-containers)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1114,7 +1123,7 @@
|
||||
"id": "9eaab1c54d66"
|
||||
},
|
||||
"source": [
|
||||
"Deploy the model to the created endpoint with the required machine-type."
|
||||
"Deploy the model to the created endpoint with the required machine type."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1167,7 +1176,7 @@
|
||||
"id": "b751978ff665"
|
||||
},
|
||||
"source": [
|
||||
"## Get explanations from the deployed model.\n",
|
||||
"## Get explanations from the deployed model\n",
|
||||
"<a name=\"section-9\"></a>\n",
|
||||
"\n",
|
||||
"For testing the deployed online model, select two instances from the test data as payload."
|
||||
@@ -1191,7 +1200,7 @@
|
||||
"id": "01532047a99e"
|
||||
},
|
||||
"source": [
|
||||
"Call the endpoint with the payload request and parse the response for explanations. The explanations consists of attributions on the independent variables used for training the model which are based on the configured attribution method. In this case, we've used the `Sampled Shapely` method which assigns credit for the outcome to each feature, and considers different permutations of the features. This method provides a sampling approximation of exact Shapley values. Further information on the attribution methods for explantions can be found at [Overview of ExplainableAI](https://cloud.google.com/vertex-ai/docs/explainable-ai/overview) page."
|
||||
"Call the endpoint with the payload request and parse the response for explanations. The explanations consists of attributions on the independent variables used for training the model which are based on the configured attribution method. In this case, we've used the `Sampled Shapely` method which assigns credit for the outcome to each feature, and considers different permutations of the features. This method provides a sampling approximation of exact Shapely values. Further information on the attribution methods for explanations can be found at [Overview of Explainable AI](https://cloud.google.com/vertex-ai/docs/explainable-ai/overview)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1266,9 +1275,9 @@
|
||||
"id": "87cf259efb64"
|
||||
},
|
||||
"source": [
|
||||
"## Next Steps\n",
|
||||
"## Next steps\n",
|
||||
"\n",
|
||||
"Since the Chicago-Taxicab dataset is continuously updating, one can preform the same kind of analysis and model training every time a new set of data is available. The date range can also be increased from a week to a month or more depending on the quality of data. Most of the steps followed in this notebook would still be valid and can be applied over the new data unless the data is too noisy. Perhaps, the notebook itself can be scheduled to run at the specified times to retrain the model using the scheduling option of the [Vertex AI workbench's Executor](https://console.cloud.google.com/vertex-ai/workbench/list/executions) feature. "
|
||||
"Since the Chicago Taxi Trips dataset is continuously updating, one can preform the same kind of analysis and model training every time a new set of data is available. The date range can also be increased from a week to a month or more depending on the quality of the data. Most of the steps followed in this notebook would still be valid and can be applied over the new data unless the data is too noisy. Perhaps, the notebook itself can be scheduled to run at the specified times to retrain the model using the scheduling option of [Vertex AI Workbench's executor](https://console.cloud.google.com/vertex-ai/workbench/list/executions). "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1277,7 +1286,7 @@
|
||||
"id": "eae8d94e3641"
|
||||
},
|
||||
"source": [
|
||||
"## Clean Up\n",
|
||||
"## Clean up\n",
|
||||
"<a name=\"section-10\"></a>\n",
|
||||
"\n",
|
||||
"Delete the resources created in this notebook.\n",
|
||||
|
||||
|
After Width: | Height: | Size: 382 KiB |
|
After Width: | Height: | Size: 445 KiB |
|
After Width: | Height: | Size: 63 KiB |
@@ -7,49 +7,53 @@
|
||||
},
|
||||
"source": [
|
||||
"# Predictive Maintenance \n",
|
||||
"\n",
|
||||
"## Table of contents\n",
|
||||
"* [Overview](#section-1)\n",
|
||||
"* [Dataset](#section-2)\n",
|
||||
"* [Objective](#section-3)\n",
|
||||
"* [Costs](#section-4)\n",
|
||||
"* [Data Analysis](#section-5)\n",
|
||||
"* [Fit a Regression model](#section-6)\n",
|
||||
"* [Data analysis](#section-5)\n",
|
||||
"* [Fit a regression model](#section-6)\n",
|
||||
"* [Evaluate the trained model](#section-7)\n",
|
||||
"* [Save the model](#section-8)\n",
|
||||
"* [Running a notebook end-to-end using **Executor**](#section-9)\n",
|
||||
"* [Running a notebook end-to-end using the executor](#section-9)\n",
|
||||
"* [Hosting the model on Vertex AI](#section-10)\n",
|
||||
" * [Create an Endpoint](#section-11)\n",
|
||||
" * [Deploy the model to the created Endpoint](#section-12)\n",
|
||||
" * [Test calling the endpoint](#section-13)\n",
|
||||
" * [Create an endpoint](#section-11)\n",
|
||||
" * [Deploy the model to the created endpoint](#section-12)\n",
|
||||
" * [Test calling the endpoint](#section-13)\n",
|
||||
"* [Clean up](#section-14)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Overview\n",
|
||||
"<a name=\"section-1\"></a>\n",
|
||||
"This notebook demonstrates performing predictive maintenance on industrial data using machine learning techniques, deploying the machine learning model on Vertex-AI and automating the workflow using executor feature of Vertex-AI.\n",
|
||||
"\n",
|
||||
"<b>Note</b>: This notebook is designed to run on managed notebooks instance of Vertex AI Workbench. Some components of this notebook may not work in other notebook environments.\n",
|
||||
"This notebook demonstrates how to perform predictive maintenance on industrial data using machine learning techniques, deploy the machine learning model on Vertex AI, and automate the workflow using the executor feature of Vertex AI Workbench.\n",
|
||||
"\n",
|
||||
"*Note: This notebook file was developed to run in a [Vertex AI Workbench managed notebooks](https://console.cloud.google.com/vertex-ai/workbench/list/managed) instance using the XGBoost (Local) kernel. Some components of this notebook may not work in other notebook environments.*\n",
|
||||
"\n",
|
||||
"## Dataset\n",
|
||||
"<a name=\"section-2\"></a>\n",
|
||||
"The dataset used in this notebook is a part of the [NASA Turbofan Engine Degradation Dataset](https://ti.arc.nasa.gov/tech/dash/groups/pcoe/prognostic-data-repository/) which consists of simulated time-series data for four sets of fleet-engines under different combinations of operational conditions and fault modes. In this notebook, only one of the engine's simulated data(FD001) has been considered to analyze and train a model that can predict the engine's remaining useful life.\n",
|
||||
"\n",
|
||||
"## Objective\n",
|
||||
"The dataset used in this notebook is a part of the [NASA Turbofan Engine Degradation Simulation dataset](https://ti.arc.nasa.gov/tech/dash/groups/pcoe/prognostic-data-repository/), which consists of simulated time-series data for four sets of fleet engines under different combinations of operational conditions and fault modes. In this notebook, only one of the engine's simulated data (FD001) has been used to analyze and train a model that can predict the engine's remaining useful life.\n",
|
||||
"\n",
|
||||
"## Objectives\n",
|
||||
"<a name=\"section-3\"></a>\n",
|
||||
"In this notebook :\n",
|
||||
"\n",
|
||||
"- Loading the required dataset from Cloud Storage bucket.\n",
|
||||
"The objectives of this notebook include:\n",
|
||||
"\n",
|
||||
"- Loading the required dataset from a Cloud Storage bucket.\n",
|
||||
"- Analyzing the fields present in the dataset.\n",
|
||||
"- Selecting the required data for the predictive maintenance model.\n",
|
||||
"- Training an XGBoost regression model for predicting the remaining useful life.\n",
|
||||
"- Evaluating the model.\n",
|
||||
"- Running the notebook end-to-end as a training job using Executor.\n",
|
||||
"- Deploying the model on Vertex-AI.\n",
|
||||
"- Deploying the model on Vertex AI.\n",
|
||||
"- Clean up.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Costs\n",
|
||||
"<a name=\"section-4\"></a>\n",
|
||||
"\n",
|
||||
"This tutorial uses the following billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
@@ -68,8 +72,10 @@
|
||||
"id": "5b15a97278df"
|
||||
},
|
||||
"source": [
|
||||
"## Kernel selection\n",
|
||||
"Select <b>XGBoost</b> kernel while running this notebook on Vertex-AIs managed instances or ensure that the following libraries are installed in the environment where this notebook is being run.\n",
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Kernel selection\n",
|
||||
"Select <b>XGBoost</b> kernel while running this notebook on Vertex AI Workbench managed notebooks instances or ensure that the following libraries are installed in the environment where this notebook is being run.\n",
|
||||
"- XGBoost\n",
|
||||
"- Pandas\n",
|
||||
"- Seaborn\n",
|
||||
@@ -80,7 +86,9 @@
|
||||
"- google.cloud.aiplatform\n",
|
||||
"- google.cloud.storage\n",
|
||||
"\n",
|
||||
"## Set your project ID"
|
||||
"### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -91,7 +99,36 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_ID = \"[your-project-id]\""
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "750bf2883c2d"
|
||||
},
|
||||
"source": [
|
||||
"Otherwise, set your project ID here."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "3c6db1ca88b9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
|
||||
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -124,11 +161,11 @@
|
||||
"id": "ea53caa30628"
|
||||
},
|
||||
"source": [
|
||||
"## Select or Create Cloud Storage Bucket for storing the model\n",
|
||||
"## Select or Create a Cloud Storage Bucket for storing the model\n",
|
||||
"\n",
|
||||
"When you create a model resource on Vertex AI using the Cloud SDK, you need to give a Cloud Storage bucket URI of the model where the model is stored. Using the model saved, you can then create Vertex AI model and endpoint resources in order to serve online predictions.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.You may also change the REGION variable, which is used for operations throughout the rest of this notebook. Make sure to choose a region where Vertex AI services are available."
|
||||
"Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. You may also change the `REGION` variable, which is used for operations throughout the rest of this notebook. Make sure to choose a region where Vertex AI services are available."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -263,7 +300,7 @@
|
||||
"id": "8cfc304d35b5"
|
||||
},
|
||||
"source": [
|
||||
"The data itself doesn't contain any feature names and thus needs its columns to be re-named. The data source already provides us with some data description. Apparently, the <b>ID</b> column represents the unit-number of the fleet-engine and <b>Cycle</b> represents the time in cycles. <b>OpSet1</b>,<b>Opset2</b> & <b>Opset3</b> represent the three operational settings that are described in the original data source and have a substantial effect on engine performance. The rest of the fields show sensor readings collected from 21 different sensors."
|
||||
"The data itself doesn't contain any feature names and thus needs its columns to be renamed. The data source already provides some data description. Apparently, the <b>ID</b> column represents the unit-number of the fleet-engine and <b>Cycle</b> represents the time in cycles. <b>OpSet1</b>,<b>Opset2</b> & <b>Opset3</b> represent the three operational settings that are described in the original data source and have a substantial effect on engine performance. The rest of the fields show sensor readings collected from 21 different sensors."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -336,7 +373,7 @@
|
||||
"id": "43c3f01352ad"
|
||||
},
|
||||
"source": [
|
||||
"On an average, there seems to be around 225 cycles per each ID in the dataset. Further, lets check the data-types of the fields and the number of null records in the data."
|
||||
"On an average, there seem to be around 225 cycles per each ID in the dataset. Next, lets check the data types of the fields and the number of null records in the data."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -418,7 +455,7 @@
|
||||
"id": "284debdf4294"
|
||||
},
|
||||
"source": [
|
||||
"Fields **SensorMeasure7**, **SensorMeasure12**, **SensorMeasure20** & **SensorMeasure21** correlate highly with many other fields. These fields can be omitted. Further, **SensorMeasure8**, **SensorMeasure11** and **SensorMeasure4** seem highly correlated with each other and so any one of them, say **SensorMeasure4** can be kept and the rest can be omitted."
|
||||
"Fields **SensorMeasure7**, **SensorMeasure12**, **SensorMeasure20** & **SensorMeasure21** correlate highly with many other fields. These fields can be omitted. Further, **SensorMeasure8**, **SensorMeasure11** and **SensorMeasure4** seem highly correlated with each other and so any one of them, for example, **SensorMeasure4**, can be kept and the rest can be omitted."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -455,7 +492,7 @@
|
||||
"id": "8197cdef2cff"
|
||||
},
|
||||
"source": [
|
||||
"As the current objective is to predict the remaining useful life(RUL) of each unit(ID), the target variable needs to be identified. Since we're dealing with a timeseries data that represents the lifetime of a unit, remaining useful life of a unit can be calculated by subtracting the current cycle from the maximum cycle of that unit.\n",
|
||||
"As the current objective is to predict the remaining useful life (RUL) of each unit (ID), the target variable needs to be identified. Since we're dealing with a timeseries data that represents the lifetime of a unit, remaining useful life of a unit can be calculated by subtracting the current cycle from the maximum cycle of that unit.\n",
|
||||
"\n",
|
||||
"\t\t\t\t\tRUL = Max. Cycle - Current Cycle \n",
|
||||
"## RUL calculation and Feature selection"
|
||||
@@ -518,7 +555,7 @@
|
||||
"id": "fc3b82355cdc"
|
||||
},
|
||||
"source": [
|
||||
"The above plot suggests that the RUL i.e., the remaining cycles is decreasing as the current cycle increases which is expected. Further, lets see the how the other fields relate to RUL in the current dataset."
|
||||
"The above plot suggests that the RUL, in other words, the remaining cycles, is decreasing as the current cycle increases which is expected. Further, lets see the how the other fields relate to RUL in the current dataset."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -557,7 +594,7 @@
|
||||
"- Fields **SensorMeasure5** and **SensorMeasure16** don't show much variance with the RUL and seem constant all the time. Hence, they can be removed.\n",
|
||||
"- Fields **SensorMeasure2**, **SensorMeasure3**, **SensorMeasure4**, **SensorMeasure13**, **SensorMeasure15** & **SensorMeasure17** show a similar rising trend.\n",
|
||||
"- **SensorMeasure9** and **SensorMeasure14** show a similar trend.\n",
|
||||
"- **SensorMeasure6** shows flatline most of the time except at a very few places and therefore can be ignored."
|
||||
"- **SensorMeasure6** shows a flatline most of the time except in a very few places and therefore can be ignored."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -583,7 +620,7 @@
|
||||
"id": "cae198bd96ef"
|
||||
},
|
||||
"source": [
|
||||
"## Split the data into Train and Test\n",
|
||||
"## Split the data into train and test\n",
|
||||
"\n",
|
||||
"Divide the dataset with the selected features into train and test sets."
|
||||
]
|
||||
@@ -613,9 +650,10 @@
|
||||
"id": "43a26d74c687"
|
||||
},
|
||||
"source": [
|
||||
"## Fit a Regression model\n",
|
||||
"## Fit a regression model\n",
|
||||
"<a name=\"section-6\"></a>\n",
|
||||
"Initialize and train a regression model using XGBoost library with the calculated RUL as the target feature."
|
||||
"\n",
|
||||
"Initialize and train a regression model using the XGBoost library with the calculated RUL as the target feature."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -769,23 +807,25 @@
|
||||
"id": "4bd88d7f4bbb"
|
||||
},
|
||||
"source": [
|
||||
"## Running a notebook end-to-end using **Executor**\n",
|
||||
"## Running a notebook end-to-end using executor\n",
|
||||
"<a name=\"section-9\"></a>\n",
|
||||
"\n",
|
||||
"### Automating the notebook execution\n",
|
||||
"All the steps followed till now can be run as a training job without using any additional code using the Notebook executor. Notebook executor can help you run a notebook file from start to end, with your choice of the environment, machine type, input parameters, and other characteristics. After setting up an execution, the notebook is executed as a job in Vertex AI custom training. Your jobs can be monitored from the Notebook Executor pane in the menu on the left.\n",
|
||||
"All the steps followed until now can be run as a training job without using any additional code using the Vertex AI Workbench executor. The executor can help you run a notebook file from start to end, with your choice of the environment, machine type, input parameters, and other characteristics. After setting up an execution, the notebook is executed as a job in Vertex AI custom training. Your jobs can be monitored from the Executor pane in the left sidebar.\n",
|
||||
"\n",
|
||||
"<img src=\"images/executor.PNG\">\n",
|
||||
"\n",
|
||||
"Executor also lets you choose the environment and machine type while automating the runs similar to Vertex AI training jobs without switching to the training jobs UI. Apart from the custom container that replicates the existing kernel by default, pre-built environments like TensorFlow Enterprise, PyTorch, and others can also be selected to run the notebook. Furthermore the required compute power can be specified by choosing from the list of machine types available, including GPUs.\n",
|
||||
"The executor also lets you choose the environment and machine type while automating the runs similar to Vertex AI training jobs without switching to the training jobs UI. Apart from the custom container that replicates the existing kernel by default, pre-built environments like TensorFlow Enterprise, PyTorch, and others can also be selected to run the notebook. The required compute power can be specified by choosing from the list of machine types available, including GPUs.\n",
|
||||
"\n",
|
||||
"## Scheduled runs on executor\n",
|
||||
"\n",
|
||||
"Notebook runs can also be scheduled recurringly with the executor. To do so, select Schedule-based recurring executions as the run type instead of One-time execution. The frequency of the job and the time when it executes is provided when you create the execution.\n",
|
||||
"\n",
|
||||
"<img src=\"https://storage.googleapis.com/gweb-cloudblog-publish/images/7_Vertex_AI_Workbench.max-1100x1100.jpg\">\n",
|
||||
"\n",
|
||||
"## Parameterizing the variables\n",
|
||||
"Executor lets you run a notebook with different sets of input parameters.If you’ve added parameter tags to any of your notebook cells, you can pass in your parameter values to the executor. More about how to use this feature can be found on this [blog](https://cloud.google.com/blog/products/ai-machine-learning/schedule-and-execute-notebooks-with-vertex-ai-workbench).\n",
|
||||
"\n",
|
||||
"The executor lets you run a notebook with different sets of input parameters. If you’ve added parameter tags to any of your notebook cells, you can pass in your parameter values to the executor. More about how to use this feature can be found on this [blog](https://cloud.google.com/blog/products/ai-machine-learning/schedule-and-execute-notebooks-with-vertex-ai-workbench).\n",
|
||||
"\n",
|
||||
"<img src=\"https://storage.googleapis.com/gweb-cloudblog-publish/images/6_Vertex_AI_Workbench.max-700x700.jpg\">\n"
|
||||
]
|
||||
@@ -944,7 +984,7 @@
|
||||
"## Clean up\n",
|
||||
"<a name=\"section-14\"></a>\n",
|
||||
"\n",
|
||||
"Undeploy the model from endpoint."
|
||||
"Undeploy the model from the endpoint."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1022,7 +1062,7 @@
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"name": "Predictive_maintenance_usecase.ipynb",
|
||||
"name": "predictive_maintenance_usecase.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
@@ -0,0 +1,823 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d1cc1c1fa076"
|
||||
},
|
||||
"source": [
|
||||
"# Pricing Optimization \n",
|
||||
"## Table of contents\n",
|
||||
"* [Overview](#section-1)\n",
|
||||
"* [Dataset](#section-2)\n",
|
||||
"* [Objective](#section-3)\n",
|
||||
"* [Costs](#section-4)\n",
|
||||
"* [Create a BigQuery dataset](#section-5)\n",
|
||||
"* [Load the dataset from Cloud Storage](#section-6)\n",
|
||||
"* [Data analysis](#section-7)\n",
|
||||
"* [Preprocess the data for training](#section-8)\n",
|
||||
"* [Train the model using BigQuery ML](#section-9)\n",
|
||||
"* [Generate forecasts from the model](#section-10)\n",
|
||||
"* [Interpret the results to choose the best price](#section-11)\n",
|
||||
"* [Clean up](#section-12)\n",
|
||||
"\n",
|
||||
"## Overview\n",
|
||||
"<a name=\"section-1\"></a>\n",
|
||||
"\n",
|
||||
"This notebook demonstrates analysis of pricing optimization on [CDM Pricing Data](https://github.com/trifacta/trifacta-google-cloud/tree/main/design-pattern-pricing-optimization) and automating the workflow using Vertex AI Workbench managed notebooks.\n",
|
||||
"\n",
|
||||
"*Note: This notebook file was developed to run in a [Vertex AI Workbench managed notebooks](https://console.cloud.google.com/vertex-ai/workbench/list/managed) instance using the Python (Local) kernel. Some components of this notebook may not work in other notebook environments.*\n",
|
||||
"\n",
|
||||
"## Dataset\n",
|
||||
"<a name=\"section-2\"></a>\n",
|
||||
"\n",
|
||||
"The dataset used in this notebook is a part of the [CDM Pricing dataset](https://github.com/trifacta/trifacta-google-cloud/blob/main/design-pattern-pricing-optimization/CDM_Pricing_large_table.csv), which consists of product sales information on specified dates.\n",
|
||||
"\n",
|
||||
"## Objective\n",
|
||||
"<a name=\"section-3\"></a>\n",
|
||||
"\n",
|
||||
"The objective of this notebook is to build a pricing optimization model using Vertex AI. The following steps have been followed: \n",
|
||||
"\n",
|
||||
"- Load the required dataset from a Cloud Storage bucket.\n",
|
||||
"- Analyze the fields present in the dataset.\n",
|
||||
"- Process the data to build a model.\n",
|
||||
"- Build a BigQuery ML forecast model on the processed data.\n",
|
||||
"- Get forecasted values from the BigQuery ML model.\n",
|
||||
"- Interpret the forecasts to identify the best prices.\n",
|
||||
"- Clean up.\n",
|
||||
"\n",
|
||||
"## Costs\n",
|
||||
"<a name=\"section-4\"></a>\n",
|
||||
"\n",
|
||||
"This tutorial uses the following billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- BigQuery\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing), [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5ed1f5e85640"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "c3f30148b66d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "750bf2883c2d"
|
||||
},
|
||||
"source": [
|
||||
"Otherwise, set your project ID here."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "3c6db1ca88b9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
|
||||
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2a1c270c7d34"
|
||||
},
|
||||
"source": [
|
||||
"### Import the required libraries and define constants\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "acc6fac1fa55"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import pandas as pd\n",
|
||||
"import seaborn as sns\n",
|
||||
"from google.cloud import bigquery\n",
|
||||
"from google.cloud.bigquery import Client"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a06006dff8f9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DATASET = \"[your-bigquery-dataset-id]\" # set the BigQuery dataset-id\n",
|
||||
"TRAINING_DATA_TABLE = \"[your-bigquery-table-id-to-store-the-training-data]\" # set the BigQuery table-id to store the training data"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "016c3d47cc69"
|
||||
},
|
||||
"source": [
|
||||
"## Create a BigQuery dataset\n",
|
||||
"<a name=\"section-5\"></a>\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "12ccd8d7956e"
|
||||
},
|
||||
"source": [
|
||||
"#@bigquery\n",
|
||||
"-- create a dataset in BigQuery\n",
|
||||
"\n",
|
||||
"CREATE SCHEMA pricing_optimization\n",
|
||||
"OPTIONS(\n",
|
||||
" location=\"us\"\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "c106b978a79b"
|
||||
},
|
||||
"source": [
|
||||
"## Load the dataset from Cloud Storage\n",
|
||||
"<a name=\"section-6\"></a>\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "8aeae9da9796"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DATA_LOCATION = \"gs://cloud-samples-data/ai-platform-unified/datasets/tabular/cdm_pricing_large_table.csv\"\n",
|
||||
"df = pd.read_csv(DATA_LOCATION)\n",
|
||||
"print(df.shape)\n",
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "7b98d5f09842"
|
||||
},
|
||||
"source": [
|
||||
"You will build a forecast model on this data and thus determine the best price for a product. For this type of model, you will not be using many fields: only the sales and price related ones. For the current execrcise, focus on the following fields:\n",
|
||||
"\n",
|
||||
"- `Product_ID`\n",
|
||||
"- `Customer_Hierarchy`\n",
|
||||
"- `Fiscal_Date`\n",
|
||||
"- `List_Price_Converged`\n",
|
||||
"- `Invoiced_quantity_in_Pieces`\n",
|
||||
"- `Net_Sales`\n",
|
||||
"\n",
|
||||
"## Data Analysis\n",
|
||||
"<a name=\"section-7\"></a>\n",
|
||||
"\n",
|
||||
"First, explore the data and distributions.\n",
|
||||
"\n",
|
||||
"Select the required columns from the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "af4b41c5eb1f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"id_col = \"Product_ID\"\n",
|
||||
"date_col = \"Fiscal_Date\"\n",
|
||||
"categ_cols = [\"Customer_Hierarchy\"]\n",
|
||||
"num_cols = [\"List_Price_Converged\", \"Invoiced_quantity_in_Pieces\", \"Net_Sales\"]\n",
|
||||
"\n",
|
||||
"df = df[[id_col, date_col] + categ_cols + num_cols].copy()\n",
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3d780043ee5b"
|
||||
},
|
||||
"source": [
|
||||
"Check the column types and null values in the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f54c445a1288"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df.info()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cd817b414c4d"
|
||||
},
|
||||
"source": [
|
||||
"This data description reveals that there are no null values in the data. Also, the field `Fiscal_Date` which is a date field is loaded as an object type. \n",
|
||||
"\n",
|
||||
"Change the type of the date field to datetime."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b160fac085c8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df[\"Fiscal_Date\"] = pd.to_datetime(df[\"Fiscal_Date\"], infer_datetime_format=True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fb4778578064"
|
||||
},
|
||||
"source": [
|
||||
"Plot the distributions for the categorical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "dd0467cd57c3"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in categ_cols:\n",
|
||||
" df[i].value_counts(normalize=True).plot(kind=\"bar\")\n",
|
||||
" plt.title(i)\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "145deed255e0"
|
||||
},
|
||||
"source": [
|
||||
"Plot the distributions for the numerical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f934137c6d82"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in num_cols:\n",
|
||||
" _, ax = plt.subplots(1, 2, figsize=(10, 4))\n",
|
||||
" df[i].plot(kind=\"box\", ax=ax[0])\n",
|
||||
" df[i].plot(kind=\"hist\", ax=ax[1])\n",
|
||||
" ax[0].set_title(i + \"-Boxplot\")\n",
|
||||
" ax[1].set_title(i + \"-Histogram\")\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f9b9c2e58380"
|
||||
},
|
||||
"source": [
|
||||
"Check the maximum date and minimum date in Fiscal_Date column."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2a10aa689f9d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"print(df[\"Fiscal_Date\"].max())\n",
|
||||
"print(df[\"Fiscal_Date\"].min())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "4834f63e2e59"
|
||||
},
|
||||
"source": [
|
||||
"Check the product distribution across each category."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4664877f5304"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"grp_cols = [\"Customer_Hierarchy\", \"Product_ID\"]\n",
|
||||
"grp_df = df[grp_cols].groupby(by=grp_cols).count().reset_index()\n",
|
||||
"grp_df.groupby(\"Customer_Hierarchy\").nunique()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "01ed02b9c8fd"
|
||||
},
|
||||
"source": [
|
||||
"Check the percentage changes in the orders based on the percentage changes in the price."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "0b2c428cb135"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# aggregate the data\n",
|
||||
"df_aggr = (\n",
|
||||
" df.groupby([\"Product_ID\", \"List_Price_Converged\"])\n",
|
||||
" .agg({\"Fiscal_Date\": min, \"Invoiced_quantity_in_Pieces\": sum, \"Net_Sales\": sum})\n",
|
||||
" .reset_index()\n",
|
||||
")\n",
|
||||
"# rename the aggregated columns\n",
|
||||
"df_aggr.rename(\n",
|
||||
" columns={\n",
|
||||
" \"Fiscal_Date\": \"First_price_date\",\n",
|
||||
" \"Invoiced_quantity_in_Pieces\": \"Total_ordered_pieces\",\n",
|
||||
" \"Net_Sales\": \"Total_net_sales\",\n",
|
||||
" },\n",
|
||||
" inplace=True,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# sort values chronologically\n",
|
||||
"df_aggr.sort_values(by=[\"Product_ID\", \"First_price_date\"], inplace=True)\n",
|
||||
"df_aggr.reset_index(drop=True, inplace=True)\n",
|
||||
"\n",
|
||||
"# add columns for previous values\n",
|
||||
"df_aggr[\"Previous_List\"] = df_aggr.groupby([\"Product_ID\"])[\n",
|
||||
" \"List_Price_Converged\"\n",
|
||||
"].shift()\n",
|
||||
"df_aggr[\"Previous_Total_ordered_pieces\"] = df_aggr.groupby([\"Product_ID\"])[\n",
|
||||
" \"Total_ordered_pieces\"\n",
|
||||
"].shift()\n",
|
||||
"\n",
|
||||
"# average price change across sku's\n",
|
||||
"df_aggr[\"price_change_perc\"] = (\n",
|
||||
" (df_aggr[\"List_Price_Converged\"] - df_aggr[\"Previous_List\"])\n",
|
||||
" / df_aggr[\"Previous_List\"].fillna(0)\n",
|
||||
" * 100\n",
|
||||
")\n",
|
||||
"df_aggr[\"order_change_perc\"] = (\n",
|
||||
" (df_aggr[\"Total_ordered_pieces\"] - df_aggr[\"Previous_Total_ordered_pieces\"])\n",
|
||||
" / df_aggr[\"Previous_Total_ordered_pieces\"].fillna(0)\n",
|
||||
" * 100\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# plot a scatterplot to visualize the changes\n",
|
||||
"sns.scatterplot(\n",
|
||||
" x=\"price_change_perc\",\n",
|
||||
" y=\"order_change_perc\",\n",
|
||||
" data=df_aggr,\n",
|
||||
" hue=\"Product_ID\",\n",
|
||||
" legend=False,\n",
|
||||
")\n",
|
||||
"plt.title(\"Percentage of change in price vs order\")\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "8259e916fe25"
|
||||
},
|
||||
"source": [
|
||||
"For most of the products, the percentage change in orders are high where the percentage changes in the prices are low. This suggests that too much change in the prices can affect the number of orders. \n",
|
||||
"\n",
|
||||
"**Note**: There seem to be some outliers in the data as percentage changes greater than 800 are found. In the current exercise, do not take any manual measures to deal with outliers as you will create a BigQuery ML timeseries model that already deals with outliers.\n",
|
||||
"\n",
|
||||
"## Preprocess the data for training\n",
|
||||
"<a name=\"section-8\"></a>\n",
|
||||
"\n",
|
||||
"Check which `Product_ID`'s have the maximum orders."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f5cbc7709c6a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_orders = df.groupby([\"Product_ID\", \"Customer_Hierarchy\"], as_index=False)[\n",
|
||||
" \"Invoiced_quantity_in_Pieces\"\n",
|
||||
"].sum()\n",
|
||||
"df_orders.loc[\n",
|
||||
" df_orders.groupby(\"Customer_Hierarchy\")[\"Invoiced_quantity_in_Pieces\"].idxmax()\n",
|
||||
"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fd6d227e513e"
|
||||
},
|
||||
"source": [
|
||||
"From the above result, you can infer the following:\n",
|
||||
"\n",
|
||||
"- Under the **Food** category, **SKU 62** has the maximum orders.\n",
|
||||
"- Under the **Manufacturing** category, **SKU 17** has the maximum orders.\n",
|
||||
"- Under the **Paper** category, **SKU 107** has the maximum orders.\n",
|
||||
"- Under the **Publishing** category, **SKU 8** has the maximum orders.\n",
|
||||
"- Under the **Utilities** category, **SKU 140** has the maximum orders.\n",
|
||||
"\n",
|
||||
"Given that there are too many ids and only a few records for most of them, consider only the above `Product_ID`s for which there are a maximum number of orders. \n",
|
||||
"\n",
|
||||
"**Note**: The `Invoiced_quantity_in_Pieces` field seems to be a *float* type rather than an *int* type as it should be. This could be because the data itself might be averaged in the first place."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2dbc0d64d157"
|
||||
},
|
||||
"source": [
|
||||
"Check the various prices available for these `Product_ID`s."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "acc1dbd2d838"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_type_food = df[(df[\"Product_ID\"] == \"SKU 62\") & (df[\"Customer_Hierarchy\"] == \"Food\")]\n",
|
||||
"print(\"Food :\")\n",
|
||||
"print(df_type_food[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_manuf = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 17\") & (df[\"Customer_Hierarchy\"] == \"Manufacturing\")\n",
|
||||
"]\n",
|
||||
"print(\"Manufacturing :\")\n",
|
||||
"print(df_type_manuf[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_paper = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 107\") & (df[\"Customer_Hierarchy\"] == \"Paper\")\n",
|
||||
"]\n",
|
||||
"print(\"Paper :\")\n",
|
||||
"print(df_type_paper[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_pub = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 8\") & (df[\"Customer_Hierarchy\"] == \"Publishing\")\n",
|
||||
"]\n",
|
||||
"print(\"Publishing :\")\n",
|
||||
"print(df_type_pub[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_util = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 140\") & (df[\"Customer_Hierarchy\"] == \"Utilities\")\n",
|
||||
"]\n",
|
||||
"print(\"Utilities :\")\n",
|
||||
"print(df_type_util[\"List_Price_Converged\"].value_counts())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f023af578c0f"
|
||||
},
|
||||
"source": [
|
||||
"In the publishing category, `Product_ID` `SKU 8` and `SKU 17` are less than or equal to two different prices in the entire data and so you will exclude them and consider the rest for building the forecast model. The idea here is to train a forecast model on the timeseries data for products with different prices.\n",
|
||||
"\n",
|
||||
"Join the data for all the `Product_ID`s into one dataframe and remove duplicate records."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a44771cc4c20"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_final = pd.concat([df_type_food, df_type_paper, df_type_util])\n",
|
||||
"df_final = (\n",
|
||||
" df_final[\n",
|
||||
" [\n",
|
||||
" \"Product_ID\",\n",
|
||||
" \"Fiscal_Date\",\n",
|
||||
" \"Customer_Hierarchy\",\n",
|
||||
" \"List_Price_Converged\",\n",
|
||||
" \"Invoiced_quantity_in_Pieces\",\n",
|
||||
" ]\n",
|
||||
" ]\n",
|
||||
" .drop_duplicates()\n",
|
||||
" .reset_index(drop=True)\n",
|
||||
")\n",
|
||||
"df_final.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "add5063df368"
|
||||
},
|
||||
"source": [
|
||||
"Save the data to a BigQuery table."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "fd82ba56571f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bq_client = bigquery.Client(project=PROJECT_ID)\n",
|
||||
"\n",
|
||||
"job_config = bigquery.LoadJobConfig(\n",
|
||||
" # Specify a (partial) schema. All columns are always written to the\n",
|
||||
" # table. The schema is used to assist in data type definitions.\n",
|
||||
" schema=[\n",
|
||||
" bigquery.SchemaField(\"Product_ID\", bigquery.enums.SqlTypeNames.STRING),\n",
|
||||
" bigquery.SchemaField(\"Fiscal_Date\", bigquery.enums.SqlTypeNames.DATE),\n",
|
||||
" bigquery.SchemaField(\"List_Price_Converged\", bigquery.enums.SqlTypeNames.FLOAT),\n",
|
||||
" bigquery.SchemaField(\n",
|
||||
" \"Invoiced_quantity_in_Pieces\", bigquery.enums.SqlTypeNames.FLOAT\n",
|
||||
" ),\n",
|
||||
" ],\n",
|
||||
" # Optionally, set the write disposition. BigQuery appends loaded rows\n",
|
||||
" # to an existing table by default, but with WRITE_TRUNCATE write\n",
|
||||
" # disposition it replaces the table with the loaded data.\n",
|
||||
" write_disposition=\"WRITE_TRUNCATE\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# save the dataframe to a table in the created dataset\n",
|
||||
"job = bq_client.load_table_from_dataframe(\n",
|
||||
" df_final,\n",
|
||||
" \"{}.{}.{}\".format(PROJECT_ID, DATASET, TRAINING_DATA_TABLE),\n",
|
||||
" job_config=job_config,\n",
|
||||
") # Make an API request.\n",
|
||||
"job.result() # Wait for the job to complete."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fca77641b03b"
|
||||
},
|
||||
"source": [
|
||||
"# Train the model using BigQuery ML\n",
|
||||
"<a name=\"section-9\"></a>\n",
|
||||
"\n",
|
||||
"Train an [Arima-Plus](https://cloud.google.com/bigquery-ml/docs/reference/standard-sql/bigqueryml-syntax-create-time-series) model on the data using BigQuery ML."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cded27507891"
|
||||
},
|
||||
"source": [
|
||||
"#@bigquery\n",
|
||||
"create or replace model pricing_optimization.bqml_arima\n",
|
||||
"options\n",
|
||||
" (model_type = 'ARIMA_PLUS',\n",
|
||||
" time_series_timestamp_col = 'Fiscal_Date',\n",
|
||||
" time_series_data_col = 'Invoiced_quantity_in_Pieces',\n",
|
||||
" time_series_id_col = 'ID'\n",
|
||||
" ) as\n",
|
||||
"select\n",
|
||||
" Fiscal_Date,\n",
|
||||
" Concat(Product_ID,\"_\" ,Cast(List_Price_Converged as string)) as ID,\n",
|
||||
" Invoiced_quantity_in_Pieces\n",
|
||||
"from\n",
|
||||
" pricing_optimization.TRAINING_DATA\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "332fd11ff32b"
|
||||
},
|
||||
"source": [
|
||||
"## Generate forecasts from the model\n",
|
||||
"<a name=\"section-10\"></a>\n",
|
||||
"\n",
|
||||
"Predict the sales for the next 30 days for each id and save to a dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ef926cdbf28e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"client = Client()\n",
|
||||
"\n",
|
||||
"query = '''\n",
|
||||
"DECLARE HORIZON STRING DEFAULT \"30\"; #number of values to forecast\n",
|
||||
"DECLARE CONFIDENCE_LEVEL STRING DEFAULT \"0.90\"; ## required confidence level\n",
|
||||
"\n",
|
||||
"EXECUTE IMMEDIATE format(\"\"\"\n",
|
||||
" SELECT\n",
|
||||
" *\n",
|
||||
" FROM \n",
|
||||
" ML.FORECAST(MODEL pricing_optimization.bqml_arima, \n",
|
||||
" STRUCT(%s AS horizon, \n",
|
||||
" %s AS confidence_level)\n",
|
||||
" )\n",
|
||||
" \"\"\",HORIZON,CONFIDENCE_LEVEL)'''\n",
|
||||
"job = client.query(query)\n",
|
||||
"dfforecast = job.to_dataframe()\n",
|
||||
"dfforecast.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "608c7de72dae"
|
||||
},
|
||||
"source": [
|
||||
"## Interpret the results to choose the best price\n",
|
||||
"<a name=\"section-11\"></a>\n",
|
||||
"\n",
|
||||
"Calculate average forecast values for the forecast duration."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e1e193680400"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dfforecast_avg = (\n",
|
||||
" dfforecast[[\"ID\", \"forecast_value\"]].groupby(\"ID\", as_index=False).mean()\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5ce395d652a3"
|
||||
},
|
||||
"source": [
|
||||
"Extract the ID and Price fields from the ID field."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "452c56fa58ed"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dfforecast_avg[\"Product_ID\"] = dfforecast_avg[\"ID\"].apply(lambda x: x.split(\"_\")[0])\n",
|
||||
"dfforecast_avg[\"Price\"] = dfforecast_avg[\"ID\"].apply(lambda x: x.split(\"_\")[1])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3cee67f4028f"
|
||||
},
|
||||
"source": [
|
||||
"Plot the average forecasted sales vs. the price of the product."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "fb351c8f383d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in dfforecast_avg[\"Product_ID\"].unique():\n",
|
||||
" dfforecast_avg[dfforecast_avg[\"Product_ID\"] == i].set_index(\"Price\").sort_values(\n",
|
||||
" \"forecast_value\"\n",
|
||||
" ).plot(kind=\"bar\")\n",
|
||||
" plt.title(\"Price vs. Average Sales for \" + i)\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "67ff3acc74a5"
|
||||
},
|
||||
"source": [
|
||||
"Based on the plots for price vs. the average forecasted orders, it can be said that to use the maximum orders, each of the considered `Product_ID`s can follow the below prices:\n",
|
||||
"\n",
|
||||
"- SKU 107's price range can be from 4.44 - 4.73 units\n",
|
||||
"- SKU 140's price can be 1.95 units\n",
|
||||
"- SKU 62's price can be 4.23 units\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Clean Up\n",
|
||||
"<a name=\"section-12\"></a>\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial. The following code deletes the entire dataset."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "d78908b8134d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Construct a BigQuery client object.\n",
|
||||
"client = bigquery.Client()\n",
|
||||
"\n",
|
||||
"# TODO(developer): Set model_id to the ID of the model to fetch.\n",
|
||||
"dataset_id = \"{PROJECT}.{DATASET}\".format(PROJECT=PROJECT_ID, DATASET=DATASET)\n",
|
||||
"\n",
|
||||
"# Use the delete_contents parameter to delete a dataset and its contents.\n",
|
||||
"# Use the not_found_ok parameter to not receive an error if the dataset has already been deleted.\n",
|
||||
"client.delete_dataset(\n",
|
||||
" dataset_id, delete_contents=True, not_found_ok=True\n",
|
||||
") # Make an API request.\n",
|
||||
"\n",
|
||||
"print(\"Deleted dataset '{}'.\".format(dataset_id))"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"name": "pricing-optimization.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -459,7 +459,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil cp gs://cloud-samples-data/ai-platform-unified/matching_engine/glove-100-angular.hdf5 ."
|
||||
"! gsutil cp gs://cloud-samples-data/vertex-ai/matching_engine/glove-100-angular.hdf5 ."
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -12,7 +12,7 @@ The purpose of this set of notebooks and markdown files is to demonstrate Google
|
||||
2. [Experimentation](stage2)
|
||||
3. [Formalization](stage3)
|
||||
4. [Evaluation](stage4)
|
||||
5. Deployment
|
||||
6. Serving
|
||||
5. [Deployment](stage5)
|
||||
6. [Serving](stage6)
|
||||
7. Monitoring
|
||||
8. Continuous Training
|
||||
|
||||
@@ -30,10 +30,78 @@ The first stage in MLOps is the collection and preparation for the purpose of de
|
||||
|
||||
[Get Started with BQ datasets](get_started_bq_datasets.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Vertex AI `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.
|
||||
- Extract a copy of the dataset from `BigQuery` to a CSV file in Cloud Storage -- compatible for `AutoML` or custom training.
|
||||
- Select rows from a `BigQuery` dataset into a `pandas` dataframe -- compatible for custom training.
|
||||
- Select rows from a `BigQuery` dataset into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.
|
||||
- Select rows from extracted CSV files into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.
|
||||
- Create a `BigQuery` dataset from CSV files.
|
||||
- Extract data from `BigQuery` table into a `DMatrix` -- compatible for custom training `XGBoost` models.
|
||||
```
|
||||
|
||||
[Get Started with Vertex datasets](get_started_vertex_datasets.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Vertex AI `Dataset` resource for:
|
||||
- image data
|
||||
- text data
|
||||
- video data
|
||||
- tabular data
|
||||
- forecasting data
|
||||
|
||||
|
||||
- Search `Dataset` resources using a filter.
|
||||
- Read a sample of a `BigQuery` dataset into a dataframe.
|
||||
- Generate statistics and data schema using TensorFlow Data Validation from the samples in the dataframe.
|
||||
- Detect anomalies in new data using TensorFlow Data Validation.
|
||||
- Generate a TFRecord feature specification using TensorFlow Transform from the data schema.
|
||||
- Export a dataset and convert to TFRecords.
|
||||
```
|
||||
|
||||
[Get Started with Dataflow](get_started_dataflow.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Offline preprocessing of data:
|
||||
- Serially - w/o dataflow
|
||||
- Parallel - with dataflow
|
||||
- Upstream preprocessing of data:
|
||||
- tabular data
|
||||
- image data
|
||||
```
|
||||
|
||||
[Get Started with Data Labeling](get_started_data_labeling.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Specialist Pool for data labelers.
|
||||
- Create a data labeling job.
|
||||
- Submit the data labeling job.
|
||||
- List data labeling jobs.
|
||||
- Cancel a data labeling job.
|
||||
```
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 1: Data Management](mlops_data_management.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Explore and visualize the data.
|
||||
- Create a Vertex AI `Dataset` resource from `BigQuery` table -- for AutoML training.
|
||||
- Extract a copy of the dataset to a CSV file in Cloud Storage.
|
||||
- Create a Vertex AI `Dataset` resource from CSV files -- alternative for AutoML training.
|
||||
- Read a sample of the `BigQuery` dataset into a dataframe.
|
||||
- Generate statistics and data schema using TensorFlow Data Validation from the samples in the dataframe.
|
||||
- Generate a TFRecord feature specification using TensorFlow Data Validation from the data schema.
|
||||
- Preprocess a portion of the BigQuery data using `Dataflow` -- for custom training.
|
||||
```
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -38,9 +38,15 @@
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -67,7 +73,7 @@
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
"The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). In this version of the dataset you consider the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -104,7 +110,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with structured (tabular) data in BigQuery:\n",
|
||||
"When doing E2E MLOps on Google Cloud, following are the best practices when dealing with structured (tabular) data in BigQuery:\n",
|
||||
"\n",
|
||||
"- For AutoML training:\n",
|
||||
" - Create a managed dataset with Vertex AI `TabularDataset`.\n",
|
||||
@@ -124,7 +130,7 @@
|
||||
" - Within the generator (upstream)\n",
|
||||
" - Within the model (downstream)\n",
|
||||
" - XGBoost model training:\n",
|
||||
" - Use BigQuery ML builtin XGBoost training.\n",
|
||||
" - Use BigQuery ML built-in XGBoost training.\n",
|
||||
" - Alternatively, create a DMatrix generator from CSV files extracted from BigQuery table.\n",
|
||||
" - Pytorch model training:\n",
|
||||
" - Extract the BigQuery to a pandas dataframe.\n",
|
||||
@@ -132,10 +138,19 @@
|
||||
" - Create a DataLoader generator from the pandas dataframe.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"- Alternately:\n",
|
||||
"- Alternatively:\n",
|
||||
" - Extract the BigQuery table to CSV files.\n",
|
||||
" - Preprocess the CSV files.\n",
|
||||
" - Create a tf.data.Dataset generator from the CSV files."
|
||||
" - Create a tf.data.Dataset generator from the CSV files.\n",
|
||||
" \n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"- BigQuery\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -146,7 +161,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages to execute this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -157,40 +172,22 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "install_xgboost"
|
||||
},
|
||||
"source": [
|
||||
"Install the latest GA version of *XGBoost* library as well."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "install_xgboost"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install -U xgboost $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"# Install the packages\n",
|
||||
"! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
"! pip3 install -U xgboost $USER_FLAG\n",
|
||||
"! pip3 install -U tensorflow-io==0.18 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -222,6 +219,39 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "a47846030fef"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "84cd83853240"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, BigQuery, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,bigquery,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -298,7 +328,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -325,6 +358,66 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "77c385f0db59"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "535223fa4b84"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebooks, then don't execute this code\n",
|
||||
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -335,12 +428,7 @@
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"When you submit a custom training job using the Vertex SDK, you upload a Python package\n",
|
||||
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
|
||||
"the code from this package. In this tutorial, Vertex AI also saves the\n",
|
||||
"trained model that results from your job in the same bucket. You can then\n",
|
||||
"create an `Endpoint` resource based on this output in order to serve\n",
|
||||
"online predictions.\n",
|
||||
"When you create a dataset resource using the Vertex SDK, you can provide a Cloud Storage bucket that contains the data. Vertex AI creates the dataset resource from the data. In this tutorial, Vertex AI also creates a dataset resource from your data in the Cloud Storage bucket.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
|
||||
]
|
||||
@@ -353,7 +441,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -364,8 +452,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -385,7 +473,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -405,7 +493,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -414,9 +502,6 @@
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
@@ -428,75 +513,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"source": [
|
||||
"#### Import BigQuery\n",
|
||||
"\n",
|
||||
"Import the BigQuery package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aiplatform\n",
|
||||
"import pandas as pd\n",
|
||||
"import xgboost as xgb\n",
|
||||
"from google.cloud import bigquery"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_xgboost"
|
||||
},
|
||||
"source": [
|
||||
"#### Import XGBoost\n",
|
||||
"\n",
|
||||
"Import the XGBoost package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_xgboost"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import xgboost as xgb"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_pandas"
|
||||
},
|
||||
"source": [
|
||||
"#### Import pandas\n",
|
||||
"\n",
|
||||
"Import the pandas package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_pandas"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import pandas as pd"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -516,7 +538,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION)"
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -538,7 +560,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bqclient = bigquery.Client()"
|
||||
"bqclient = bigquery.Client(project=PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -549,7 +571,7 @@
|
||||
"source": [
|
||||
"#### Location of BigQuery training data.\n",
|
||||
"\n",
|
||||
"Now set the variable `IMPORT_FILE` to the location of the data table in BigQuery."
|
||||
"Now, set the variable `IMPORT_FILE` to the location of the data table in BigQuery and `BQ_TABLE` with the table id."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -591,10 +613,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
"dataset = aiplatform.TabularDataset.create(\n",
|
||||
" display_name=\"NOAA historical weather data\" + \"_\" + TIMESTAMP,\n",
|
||||
" bq_source=[IMPORT_FILE],\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_URI[5:]},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"label_column = \"mean_temp\"\n",
|
||||
@@ -610,7 +632,7 @@
|
||||
"source": [
|
||||
"### Copy the dataset to Cloud Storage\n",
|
||||
"\n",
|
||||
"Next, you make a copy of the BigQuery dataset, as a CSV file, to Cloud Storage using the BigQuery extract command.\n",
|
||||
"Next, you make a copy of the BigQuery table as a CSV file, to Cloud Storage using the BigQuery extract command.\n",
|
||||
"\n",
|
||||
"Learn more about [BigQuery command line interface](https://cloud.google.com/bigquery/docs/reference/bq-cli-reference)."
|
||||
]
|
||||
@@ -626,9 +648,9 @@
|
||||
"comps = BQ_TABLE.split(\".\")\n",
|
||||
"BQ_PROJECT_DATASET_TABLE = comps[0] + \":\" + comps[1] + \".\" + comps[2]\n",
|
||||
"\n",
|
||||
"! bq --location=us extract --destination_format CSV $BQ_PROJECT_DATASET_TABLE $BUCKET_NAME/mydata*.csv\n",
|
||||
"! bq --location=us extract --destination_format CSV $BQ_PROJECT_DATASET_TABLE $BUCKET_URI/mydata*.csv\n",
|
||||
"\n",
|
||||
"IMPORT_FILES = ! gsutil ls $BUCKET_NAME/mydata*.csv\n",
|
||||
"IMPORT_FILES = ! gsutil ls $BUCKET_URI/mydata*.csv\n",
|
||||
"\n",
|
||||
"print(IMPORT_FILES)\n",
|
||||
"\n",
|
||||
@@ -664,15 +686,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" gcs_source = IMPORT_FILES\n",
|
||||
"else:\n",
|
||||
" gcs_source = [IMPORT_FILE]\n",
|
||||
"gcs_source = IMPORT_FILES\n",
|
||||
"\n",
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
"dataset = aiplatform.TabularDataset.create(\n",
|
||||
" display_name=\"NOAA historical weather data\" + \"_\" + TIMESTAMP,\n",
|
||||
" gcs_source=gcs_source,\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_URI[5:]},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -694,6 +713,30 @@
|
||||
"Learn more about [Creating BigQuery views](https://cloud.google.com/bigquery/docs/views)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "7dc142433e50"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Set dataset name and view name in BigQuery\n",
|
||||
"BQ_MY_DATASET = \"[your-dataset-name]\"\n",
|
||||
"BQ_MY_TABLE = \"[your-view-name]\"\n",
|
||||
"\n",
|
||||
"# Otherwise, use the default names\n",
|
||||
"if (\n",
|
||||
" BQ_MY_DATASET == \"\"\n",
|
||||
" or BQ_MY_DATASET is None\n",
|
||||
" or BQ_MY_DATASET == \"[your-dataset-name]\"\n",
|
||||
"):\n",
|
||||
" BQ_MY_DATASET = \"mlops_dataset_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"if BQ_MY_TABLE == \"\" or BQ_MY_TABLE is None or BQ_MY_TABLE == \"[your-view-name]\":\n",
|
||||
" BQ_MY_TABLE = \"mlops_view_\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -702,8 +745,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BQ_MY_DATASET = 'mydataset'\n",
|
||||
"BQ_MY_TABLE = 'myview'\n",
|
||||
"# Create the resources\n",
|
||||
"! bq --location=US mk -d \\\n",
|
||||
"$PROJECT_ID:$BQ_MY_DATASET\n",
|
||||
"\n",
|
||||
@@ -744,8 +786,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download a table.\n",
|
||||
"table = bigquery.TableReference.from_string(\"bigquery-public-data.samples.gsod\")\n",
|
||||
"# Download the table.\n",
|
||||
"table = bigquery.TableReference.from_string(BQ_TABLE)\n",
|
||||
"\n",
|
||||
"rows = bqclient.list_rows(\n",
|
||||
" table,\n",
|
||||
@@ -1031,22 +1073,6 @@
|
||||
"TABLE_ID = \"gsod\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_bigquery_dataset(dataset_id):\n",
|
||||
" dataset = bigquery.Dataset(\n",
|
||||
" bigquery.dataset.DatasetReference(PROJECT_ID, dataset_id)\n",
|
||||
" )\n",
|
||||
" dataset.location = \"us\"\n",
|
||||
"\n",
|
||||
" try:\n",
|
||||
" dataset = bqclient.create_dataset(dataset) # API request\n",
|
||||
" return True\n",
|
||||
" except Exception as err:\n",
|
||||
" print(err)\n",
|
||||
" if err.code != 409: # http_client.CONFLICT\n",
|
||||
" raise\n",
|
||||
" return False\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def load_data_into_bigquery(url, dataset_id, table_id):\n",
|
||||
" create_bigquery_dataset(dataset_id)\n",
|
||||
" dataset = bqclient.dataset(dataset_id)\n",
|
||||
@@ -1079,13 +1105,11 @@
|
||||
"source": [
|
||||
"### Read BigQuery table into XGboost DMatrix\n",
|
||||
"\n",
|
||||
"Currently, there is no direct data feeding connector between BigQuery and the open source XGBoost.\n",
|
||||
"Currently, there is no direct data feeding connector between BigQuery and the open source XGBoost. The BigQuery ML service has a built-in XGBoost training module.\n",
|
||||
"\n",
|
||||
"The BigQuery ML service has XGBoost training builtin.\n",
|
||||
"Alernatively, you extract the data either as a pandas dataframe or as CSV files. The extracted data is then given as an input to a `DMatrix` object when training the model.\n",
|
||||
"\n",
|
||||
"Alernatively, you extract the data either as a pandas dataframe or as CSV files. The extracted data is then inputted to a `DMatrix` object when training the model.\n",
|
||||
"\n",
|
||||
"Learn more about [Getting started with builtin XGBoost](https://cloud.google.com/ai-platform/training/docs/algorithms/xgboost-start)"
|
||||
"Learn more about [Getting started with built-in XGBoost](https://cloud.google.com/ai-platform/training/docs/algorithms/xgboost-start)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1096,7 +1120,7 @@
|
||||
"source": [
|
||||
"### Read pandas table into XGboost DMatrix\n",
|
||||
"\n",
|
||||
"Next, you load the pandas dataframe into a `DMatrix` object. XGBoost does not support non-numeric inputs. Any column that is categorical will need to be one-hot encoded prior to loading the dataframe."
|
||||
"Next, you load the pandas dataframe into a `DMatrix` object. XGBoost does not support non-numeric inputs. Any column that is categorical need to be one-hot encoded prior to loading the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1122,7 +1146,7 @@
|
||||
"source": [
|
||||
"### Read CSV files into XGboost DMatrix\n",
|
||||
"\n",
|
||||
"Currently, there is no Cloud Storage support in XGBoost. If you use CSV files for input, you will need to download them locally."
|
||||
"Currently, there is no Cloud Storage support in XGBoost. If you use CSV files for input, you need to download them locally."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1144,87 +1168,41 @@
|
||||
"id": "cleanup:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"# Cleaning up\n",
|
||||
"# Clean up\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"- Vertex AI Dataset resource\n",
|
||||
"- Cloud Storage Bucket\n",
|
||||
"- BigQuery Dataset\n",
|
||||
"\n",
|
||||
"Set `delete_storage` to _True_ to delete the storage resources used in this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "cleanup:mbsdk"
|
||||
"id": "47ad926d84e8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the dataset using the Vertex dataset object\n",
|
||||
"dataset.delete()\n",
|
||||
"# Delete the temporary BigQuery dataset\n",
|
||||
"! bq rm -r -f $PROJECT_ID:$DATASET_ID\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"delete_storage = False\n",
|
||||
"if delete_storage or os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Delete the created GCS bucket\n",
|
||||
" ! gsutil rm -r $BUCKET_URI\n",
|
||||
" # Delete the created BigQuery datasets\n",
|
||||
" ! bq rm -r -f $PROJECT_ID:$BQ_MY_DATASET"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -676,6 +676,31 @@
|
||||
"dataframe[\"station_number\"] = pd.to_numeric(dataframe[\"station_number\"])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"source": [
|
||||
"### Create BQ dataset resource\n",
|
||||
"\n",
|
||||
"First, you create an empty dataset resource in your project."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BQ_MY_DATASET = 'samples'\n",
|
||||
"BQ_MY_TABLE = 'gsod'\n",
|
||||
"! bq --location=US mk -d \\\n",
|
||||
"$PROJECT_ID:$BQ_MY_DATASET"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
|
||||
@@ -33,18 +33,223 @@ The second stage in MLOps is experimenting in developing one or more baseline mo
|
||||
|
||||
[Get Started with Vertex Experiments and Vertex ML Metadata](get_started_vertex_experiments.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Use Python logging to log training configuration/results locally.
|
||||
- Use Google Cloud Logging to log training configuration/results in cloud storage.
|
||||
- Create a Vertex AI `Experiment` resource.
|
||||
- Instantiate an experiment run.
|
||||
- Log parameters for the run.
|
||||
- Log metrics for the run.
|
||||
- Display the logged experiment run.
|
||||
```
|
||||
|
||||
[Get Started with Vertex TensorBoard](get_started_vertex_tensorboard.ipynb)
|
||||
|
||||
[Get Started with Custom Training Packages](get_started_vertex_training.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a TensorBoard callback when training a model.
|
||||
- Using Tensorboard with locally trained model.
|
||||
- Using Vertex AI TensorBoard with Vertex AI Training.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Tensorflow)](get_started_vertex_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a single Python script.
|
||||
- Training using a Python package.
|
||||
- Training using a custom training image.
|
||||
- Laying out a training package.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Scikit-Learn)](get_started_vertex_training_sklearn.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (XGBoost)](get_started_vertex_training_xgboost.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Pytorch)](get_started_vertex_training_pytorch.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Single node training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (R)](get_started_vertex_training_r.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Locally train an R model in a notebook using %%R magic commands
|
||||
- Create a deployment image with trained R model and serving functions.
|
||||
- Test the deployment image locally.
|
||||
- Create a `Vertex AI Model` resource for the deployment image with embedded R model.
|
||||
- Deploy the deployment image with embedded R model to a `Vertex AI Endpoint` resource.
|
||||
- Test the deployment image with embedded R model.
|
||||
- Create a R-to-Python training package.
|
||||
- Create a training image for training the model.
|
||||
- Train a R model using `Vertex AI Trainingh` service with the R-to-Python training package.
|
||||
```
|
||||
|
||||
[Get Started with Distributed Training](get_started_vertex_distributed_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- `MirroredStrategy`: Train on a single VM with multiple GPUs.
|
||||
- `MultiWorkerMirroredStrategy`: Train on multiple VMs with automatic setup of replicas.
|
||||
- `MultiWorkerMirroredStrategy`: Train on multiple VMs with fine grain control of replicas.
|
||||
- `ReductionServer`: Train on multiple VMS and sync updates across VMS with `Vertex AI Reduction Server`.
|
||||
- `TPUTraining`: Train with multiple Cloud TPUs.
|
||||
```
|
||||
|
||||
[Get Started with Vizier Hyperparameter Tuning](get_started_vertex_vizier.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Hyperparameter tuning with Random algorithm.
|
||||
- Hyperparameter tuning with Vizier (Bayesian) algorithm.
|
||||
```
|
||||
|
||||
[Get Started with AutoML Training](get_started_automl_training.ipynb)
|
||||
|
||||
[Get Started with BQML Training](get_started_bqml_training.ipyn)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Train an image model.
|
||||
- Export the image model as an edge model.
|
||||
- Train a tabular model.
|
||||
- Export the tabular model as a cloud model.
|
||||
- Train a text model.
|
||||
```
|
||||
|
||||
[Get Started with BQML Training](get_started_bqml_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a local BigQuery table in your project
|
||||
- Train a BQML model
|
||||
- Evaluate the BQML model
|
||||
- Export the BQML model as a cloud model
|
||||
- Upload the exported model as a `Vertex AI Model` resource
|
||||
- Hyperparameter tune a BQML model with `Vertex AI Vizier`
|
||||
- Automatically register a BQML model to `Vertex AI Model Registry`
|
||||
```
|
||||
|
||||
[Get Started with Vertex Feature Store](get_started_vertex_feature_store.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a Vertex AI `Featurestore` resource.
|
||||
- Creating `EntityType` resources for the `Featurestore` resource.
|
||||
- Creating `Feature` resources for each `EntityType` resource.
|
||||
- Import feature values (entity data items) into `Featurestore` resource from Cloud Storage.
|
||||
- Import feature values (entity data items) into `Featurestore` resource from pandas DataFrame.
|
||||
- Perform online serving from a `Featurestore` resource.
|
||||
- Perform batch serving from a `Featurestore` resource.
|
||||
```
|
||||
|
||||
[Get Started with Google CMEK Training](get_started_with_cmek_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a customer managed encryption key.
|
||||
- Creating an image dataset with CMEK encryption.
|
||||
- Train an AutoML model with CMEK encryption.
|
||||
```
|
||||
|
||||
[Get Started with TensorFlow Hub models](get_started_with_tfhub_models.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Download a TensorFlow Hub prebuilt model.
|
||||
- Add the task component as a classifier for the CIFAR-10 dataset.
|
||||
- Fine tune locally the model with transfer learning training.
|
||||
- Construct a custom training script:
|
||||
- Get training data from TensorFlow Datasets
|
||||
- Get model architecture from TensorFlow Hub
|
||||
- Train then model
|
||||
- Save model artifacts and upload as Vertex AI Model resource.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI TabNet builtin algorithm](get_started_with_tabnet.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Get the training data.
|
||||
- Configure training parameters for the Vertex AI TabNet container.
|
||||
- Train the model using Vertex AI Training using CSV data.
|
||||
- Upload the model as a Vertex AI Model resource.
|
||||
- Deploy the Vertex AI Model resource to a Vertex AI Endpoint resource.
|
||||
- Make a prediction with the deployed model.
|
||||
- Hyperparameter tuning the Vertex AI TabNet model.
|
||||
- Train the model using Vertex AI Training using BigQuery table.
|
||||
```
|
||||
[Get Started with Vertex AI TabNet builtin algorithm](get_started_with_tabnet.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Get the training data.
|
||||
- Configure training parameters for the Vertex AI TabNet container.
|
||||
- Train the model using Vertex AI Training using CSV data.
|
||||
- Upload the model as a Vertex AI Model resource.
|
||||
- Deploy the Vertex AI Model resource to a Vertex AI Endpoint resource.
|
||||
- Make a prediction with the deployed model.
|
||||
- Hyperparameter tuning the Vertex AI TabNet model.
|
||||
- Train the model using Vertex AI Training using BigQuery table.
|
||||
```
|
||||
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 2: Experimentation](mlops_experimentation.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Review the `Dataset` resource created during stage 1.
|
||||
- Train an AutoML tabular binary classifier model in the background.
|
||||
- Build the experimental model architecture.
|
||||
- Construct a custom training package for the `Dataset` resource.
|
||||
- Test the custom training package locally.
|
||||
- Test the custom training package in the cloud with Vertex AI Training.
|
||||
- Hyperparameter tune the model training with Vertex AI Vizier.
|
||||
- Train the custom model with Vertex AI Training.
|
||||
- Add a serving function for online/batch prediction to the custom model.
|
||||
- Test the custom model with the serving function.
|
||||
- Evaluate the custom model using Vertex AI Batch Prediction
|
||||
- Wait for the AutoML training job to complete.
|
||||
- Evaluate the AutoML model using Vertex AI Batch Prediction with the same evaluation slices as the custom model.
|
||||
- Set the evaluation results of the AutoML model as the baseline.
|
||||
- If the evaluation of the custom model is below baseline, continue to experiment with the custom model.
|
||||
- If the evaluation of the custom model is above baseline, save the model as the first best model.
|
||||
```
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -38,9 +38,15 @@
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -65,9 +71,11 @@
|
||||
"id": "dataset:flowers,icn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"### Datasets\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip."
|
||||
"#### Image\n",
|
||||
"\n",
|
||||
"The image dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower in a given image from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -76,9 +84,9 @@
|
||||
"id": "dataset:gsod,lrg"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"#### Tabular\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
"The tabular dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -87,9 +95,20 @@
|
||||
"id": "dataset:happydb,tcn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"#### Text\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket."
|
||||
"The text dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "98eb93ec6faa"
|
||||
},
|
||||
"source": [
|
||||
"#### Video\n",
|
||||
"\n",
|
||||
"The video dataset used for this tutorial is the golf swing recognition portion of the [Human Motion dataset](https://todo) from [MIT](http://cbcl.mit.edu/publications/ps/Kuehne_etal_iccv11.pdf). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model will predict the start frame where a golf swing begins."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -108,11 +127,12 @@
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Train an image model.\n",
|
||||
"- Export the image model as an edge model.\n",
|
||||
"- Train a tabular model.\n",
|
||||
"- Export the tabular model as a cloud model.\n",
|
||||
"- Train a text model."
|
||||
"- Train an image model\n",
|
||||
"- Export the image model as an edge model\n",
|
||||
"- Train a tabular model\n",
|
||||
"- Export the tabular model as a cloud model\n",
|
||||
"- Train a text model\n",
|
||||
"- Train a video model"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -125,9 +145,24 @@
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are best practices for when to use AutoML:\n",
|
||||
"\n",
|
||||
"**You have a limited amount of training data**\n",
|
||||
"* **You have a limited amount of training data**\n",
|
||||
"\n",
|
||||
"**You want to establish a baseline metric before experimenting with a custom model**"
|
||||
"* **You want to establish a baseline metric before experimenting with a custom model**"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fb3451ce8e47"
|
||||
},
|
||||
"source": [
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -138,7 +173,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing the MLOps notebooks."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -149,20 +184,18 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-storage $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -200,6 +233,23 @@
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, Compute Engine and Cloud Storage APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,storage_component).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n",
|
||||
"\n",
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
@@ -270,7 +320,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -297,6 +350,63 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3ffa6b6c7cdb"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2b72272258fc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebook, then don't execute this code\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -320,7 +430,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -331,8 +442,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
|
||||
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -352,7 +464,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -372,7 +484,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -395,7 +507,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -417,7 +529,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -448,7 +560,7 @@
|
||||
"source": [
|
||||
"## AutoML image models\n",
|
||||
"\n",
|
||||
"AutoML can train the following types of models:\n",
|
||||
"AutoML can train the following types of image models:\n",
|
||||
"\n",
|
||||
"- classification\n",
|
||||
"- objection detection\n",
|
||||
@@ -504,10 +616,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" FILE = IMPORT_FILES[0]\n",
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
@@ -545,10 +654,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.ImageDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.ImageDataset.create(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
@@ -593,8 +702,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLImageTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
"dag = aiplatform.AutoMLImageTrainingJob(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" multi_label=False,\n",
|
||||
" model_type=\"MOBILE_TF_LOW_LATENCY_1\",\n",
|
||||
@@ -612,14 +721,14 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
"- `training_fraction_split`: The percentage of the dataset to use for training.\n",
|
||||
"- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n",
|
||||
"- `validation_fraction_split`: The percentage of the dataset to use for validation.\n",
|
||||
"- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n",
|
||||
"- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of milli node-hours (1000 = node-hour).\n",
|
||||
"- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n",
|
||||
"\n",
|
||||
"The `run` method when completed returns the `Model` resource.\n",
|
||||
@@ -637,7 +746,7 @@
|
||||
"source": [
|
||||
"model = dag.run(\n",
|
||||
" dataset=dataset,\n",
|
||||
" model_display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" model_display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.8,\n",
|
||||
" validation_fraction_split=0.1,\n",
|
||||
" test_fraction_split=0.1,\n",
|
||||
@@ -653,7 +762,7 @@
|
||||
},
|
||||
"source": [
|
||||
"## Review model evaluation scores\n",
|
||||
"After your model has finished training, you can review the evaluation scores for it.\n",
|
||||
"After your model training has finished, you can review the evaluation scores for it.\n",
|
||||
"\n",
|
||||
"First, you need to get a reference to the new model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project."
|
||||
]
|
||||
@@ -667,11 +776,13 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"models = aiplatform.Model.list(filter=\"display_name=flowers_\" + TIMESTAMP)\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"model_service_client = aiplatform.gapic.ModelServiceClient(\n",
|
||||
" client_options=client_options\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
@@ -721,7 +832,7 @@
|
||||
"source": [
|
||||
"### Get test item\n",
|
||||
"\n",
|
||||
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction."
|
||||
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model. You are just looking at how to make a prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -753,7 +864,7 @@
|
||||
"\n",
|
||||
"#### Request\n",
|
||||
"\n",
|
||||
"Since in this example your test item is in a Cloud Storage bucket, you open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, you encode the bytes into base64 -- which makes the content safe from modification while transmitting binary data over the network.\n",
|
||||
"Since your test item is in a public Cloud Storage bucket in this example, you copy it to your bucket and read the contents of the image using `Cloud Storage SDK`. To pass the test data to the prediction service, you encode the bytes into base64 which makes the content safe from modification while transmitting binary data over the network.\n",
|
||||
"\n",
|
||||
"The format of each instance is:\n",
|
||||
"\n",
|
||||
@@ -775,32 +886,66 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "predict_request:mbsdk,icn"
|
||||
"id": "1c1d53e89beb"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import base64\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"from google.cloud import storage\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"# Copy the test image to the Cloud storage bucket as \"test.jpg\"\n",
|
||||
"test_image_local = \"{}/test.jpg\".format(BUCKET_URI)\n",
|
||||
"! gsutil cp $test_item $test_image_local\n",
|
||||
"\n",
|
||||
"# Download the test image in bytes format\n",
|
||||
"storage_client = storage.Client(project=PROJECT_ID)\n",
|
||||
"bucket = storage_client.bucket(bucket_name=BUCKET_NAME)\n",
|
||||
"test_content = bucket.get_blob(\"test.jpg\").download_as_bytes()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"instances = [{\"content\": base64.b64encode(test_content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances)\n",
|
||||
"\n",
|
||||
"print(prediction)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3b1b67898533"
|
||||
},
|
||||
"source": [
|
||||
"#### Alternate method using [GFile](https://www.tensorflow.org/api_docs/python/tf/io/gfile/GFile)\n",
|
||||
"\n",
|
||||
"Alternatively, [GFile](https://www.tensorflow.org/api_docs/python/tf/io/gfile/GFile) method from tensorflow-io library can be used to read the data from Cloud storage directly. The following code snippet does the same :\n",
|
||||
"\n",
|
||||
"```\n",
|
||||
"import base64\n",
|
||||
"import tensorflow as tf\n",
|
||||
"\n",
|
||||
"# Read the test file using GFile\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances)\n",
|
||||
"\n",
|
||||
"print(prediction)\n",
|
||||
"```\n",
|
||||
"Nevertheless, `tf.io.gfile.GFile` supports multiple file system implementations, including local files, Google Cloud Storage (using a gs:// prefix), and HDFS (using an hdfs:// prefix)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -846,7 +991,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = model.export_model(\n",
|
||||
" artifact_destination=BUCKET_NAME, export_format_id=\"tflite\", sync=True\n",
|
||||
" artifact_destination=BUCKET_URI, export_format_id=\"tflite\", sync=True\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_package = response[\"artifactOutputUri\"]"
|
||||
@@ -987,10 +1132,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TabularDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.TabularDataset.create(\n",
|
||||
" display_name=\"gsod_\" + TIMESTAMP,\n",
|
||||
" bq_source=[IMPORT_FILE],\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME[5:]},\n",
|
||||
" labels={\"user_metadata\": BUCKET_NAME},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"label_column = \"mean_temp\"\n",
|
||||
@@ -1046,9 +1191,7 @@
|
||||
" - regression:\n",
|
||||
" - `minimize-rmse`\n",
|
||||
" - `minimize-mae`\n",
|
||||
" - `minimize-rmsle`\n",
|
||||
"\n",
|
||||
"The instantiated object is the DAG (directed acyclic graph) for the training pipeline."
|
||||
" - `minimize-rmsle`"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1059,8 +1202,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLTabularTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
"dag = aiplatform.AutoMLTabularTrainingJob(\n",
|
||||
" display_name=\"gsod_\" + TIMESTAMP,\n",
|
||||
" optimization_prediction_type=\"regression\",\n",
|
||||
" optimization_objective=\"minimize-rmse\",\n",
|
||||
" column_transformations=TRANSFORMATIONS,\n",
|
||||
@@ -1077,7 +1220,7 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
@@ -1103,7 +1246,7 @@
|
||||
"source": [
|
||||
"model = dag.run(\n",
|
||||
" dataset=dataset,\n",
|
||||
" model_display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" model_display_name=\"gsod_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.8,\n",
|
||||
" validation_fraction_split=0.1,\n",
|
||||
" test_fraction_split=0.1,\n",
|
||||
@@ -1134,11 +1277,13 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"models = aiplatform.Model.list(filter=\"display_name=gsod_\" + TIMESTAMP)\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"model_service_client = aiplatform.gapic.ModelServiceClient(\n",
|
||||
" client_options=client_options\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
@@ -1177,7 +1322,7 @@
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -1218,7 +1363,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = model.export_model(\n",
|
||||
" artifact_destination=BUCKET_NAME, export_format_id=\"tf-saved-model\", sync=True\n",
|
||||
" artifact_destination=BUCKET_URI, export_format_id=\"tf-saved-model\", sync=True\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_package = response[\"artifactOutputUri\"]"
|
||||
@@ -1350,10 +1495,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" FILE = IMPORT_FILES[0]\n",
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
@@ -1391,10 +1533,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.TextDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.TextDataset.create(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.text.single_label_classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.text.single_label_classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
@@ -1420,9 +1562,7 @@
|
||||
" - `sentiment`: A text sentiment analysis model.\n",
|
||||
" - `extraction`: A text entity extraction model.\n",
|
||||
"- `multi_label`: If a classification task, whether single (False) or multi-labeled (True).\n",
|
||||
"- `sentiment_max`: If a sentiment analysis task, the maximum sentiment value.\n",
|
||||
"\n",
|
||||
"The instantiated object is the DAG (directed acyclic graph) for the training pipeline."
|
||||
"- `sentiment_max`: If a sentiment analysis task, the maximum sentiment value.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1433,7 +1573,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLTextTrainingJob(\n",
|
||||
"dag = aiplatform.AutoMLTextTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" multi_label=False,\n",
|
||||
@@ -1450,7 +1590,7 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
@@ -1501,11 +1641,13 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"models = aiplatform.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"model_service_client = aiplatform.gapic.ModelServiceClient(\n",
|
||||
" client_options=client_options\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
@@ -1542,7 +1684,7 @@
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -1686,10 +1828,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if \"IMPORT_FILES\" in globals():\n",
|
||||
" FILE = IMPORT_FILES[0]\n",
|
||||
"else:\n",
|
||||
" FILE = IMPORT_FILE\n",
|
||||
"FILE = IMPORT_FILE\n",
|
||||
"\n",
|
||||
"count = ! gsutil cat $FILE | wc -l\n",
|
||||
"print(\"Number of Examples\", int(count[0]))\n",
|
||||
@@ -1726,10 +1865,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.VideoDataset.create(\n",
|
||||
" display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n",
|
||||
"dataset = aiplatform.VideoDataset.create(\n",
|
||||
" display_name=\"human_motion_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.video.classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.video.classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
@@ -1753,9 +1892,7 @@
|
||||
"- `prediction_type`: The type task to train the model for.\n",
|
||||
" - `classification`: A video classification model.\n",
|
||||
" - `object_tracking`: A video object tracking model.\n",
|
||||
" - `action_recognition`: A video action recognition model.\n",
|
||||
"\n",
|
||||
"The instantiated object is the DAG (directed acyclic graph) for the training pipeline."
|
||||
" - `action_recognition`: A video action recognition model."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1766,8 +1903,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dag = aip.AutoMLVideoTrainingJob(\n",
|
||||
" display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
"dag = aiplatform.AutoMLVideoTrainingJob(\n",
|
||||
" display_name=\"human_motion_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
@@ -1782,7 +1919,7 @@
|
||||
"source": [
|
||||
"#### Run the training pipeline\n",
|
||||
"\n",
|
||||
"Next, you run the DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"Next, you run the created DAG to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `dataset`: The `Dataset` resource to train the model.\n",
|
||||
"- `model_display_name`: The human readable name for the trained model.\n",
|
||||
@@ -1804,7 +1941,7 @@
|
||||
"source": [
|
||||
"model = dag.run(\n",
|
||||
" dataset=dataset,\n",
|
||||
" model_display_name=\"happydb_\" + TIMESTAMP,\n",
|
||||
" model_display_name=\"human_motion_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.8,\n",
|
||||
" test_fraction_split=0.2,\n",
|
||||
")"
|
||||
@@ -1831,11 +1968,13 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Get model resource ID\n",
|
||||
"models = aip.Model.list(filter=\"display_name=happydb_\" + TIMESTAMP)\n",
|
||||
"models = aiplatform.Model.list(filter=\"display_name=human_motion_\" + TIMESTAMP)\n",
|
||||
"\n",
|
||||
"# Get a reference to the Model Service client\n",
|
||||
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
|
||||
"model_service_client = aip.gapic.ModelServiceClient(client_options=client_options)\n",
|
||||
"model_service_client = aiplatform.gapic.ModelServiceClient(\n",
|
||||
" client_options=client_options\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model_evaluations = model_service_client.list_model_evaluations(\n",
|
||||
" parent=models[0].resource_name\n",
|
||||
@@ -1899,16 +2038,7 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1919,66 +2049,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_dataset = True\n",
|
||||
"delete_pipeline = True\n",
|
||||
"delete_model = True\n",
|
||||
"delete_endpoint = True\n",
|
||||
"delete_batchjob = True\n",
|
||||
"delete_customjob = True\n",
|
||||
"delete_hptjob = True\n",
|
||||
"delete_bucket = True\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"# Delete the dataset using the Vertex fully qualified identifier for the dataset\n",
|
||||
"try:\n",
|
||||
" if delete_dataset and \"dataset_id\" in globals():\n",
|
||||
" clients[\"dataset\"].delete_dataset(name=dataset_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n",
|
||||
"try:\n",
|
||||
" if delete_pipeline and \"pipeline_id\" in globals():\n",
|
||||
" clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the model using the Vertex fully qualified identifier for the model\n",
|
||||
"try:\n",
|
||||
" if delete_model and \"model_to_deploy_id\" in globals():\n",
|
||||
" clients[\"model\"].delete_model(name=model_to_deploy_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n",
|
||||
"try:\n",
|
||||
" if delete_endpoint and \"endpoint_id\" in globals():\n",
|
||||
" clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the batch job using the Vertex fully qualified identifier for the batch job\n",
|
||||
"try:\n",
|
||||
" if delete_batchjob and \"batch_job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the custom job using the Vertex fully qualified identifier for the custom job\n",
|
||||
"try:\n",
|
||||
" if delete_customjob and \"job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_custom_job(name=job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n",
|
||||
"try:\n",
|
||||
" if delete_hptjob and \"hpt_job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -39,8 +39,9 @@
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_bqml_training.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -67,7 +68,7 @@
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the Penguins dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset predicts the species."
|
||||
"The dataset used for this tutorial is the Penguins dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). This version of the dataset is used to predict the species of penguins from the available features like culmen-length, flipper-depth etc."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -84,16 +85,26 @@
|
||||
"\n",
|
||||
"- `BigQueryML Training`\n",
|
||||
"- `Vertex AI Model resource`\n",
|
||||
"- `Vertex AI Vizier.\n",
|
||||
"- `Vertex AI Vizier`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Create a local BQ table in your project.\n",
|
||||
"- Train a BQML model.\n",
|
||||
"- Evaluate the BQML model.\n",
|
||||
"- Export the BQML model as a cloud model.\n",
|
||||
"- Upload the exported model as a Vertex AI Model resource.\n",
|
||||
"- Hyperparameter tune a BQML model with Vertex AI Vizier."
|
||||
"- Create a local BigQuery table in your project\n",
|
||||
"- Train a BQML model\n",
|
||||
"- Evaluate the BQML model\n",
|
||||
"- Export the BQML model as a cloud model\n",
|
||||
"- Upload the exported model as a `Vertex AI Model` resource\n",
|
||||
"- Hyperparameter tune a BQML model with `Vertex AI Vizier`\n",
|
||||
"- Automatically register a BQML model to `Vertex AI Model Registry`\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"- BigQuery\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -104,7 +115,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -115,20 +126,20 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"# Install the packages\n",
|
||||
"! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-bigquery $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -236,7 +247,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -347,9 +361,6 @@
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
@@ -361,28 +372,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"source": [
|
||||
"#### Import BigQuery\n",
|
||||
"\n",
|
||||
"Import the BigQuery package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aiplatform\n",
|
||||
"from google.cloud import bigquery"
|
||||
]
|
||||
},
|
||||
@@ -392,7 +382,7 @@
|
||||
"id": "init_aip:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"### Initialize Vertex AI and BigQuery SDKs for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
@@ -405,7 +395,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -414,8 +404,6 @@
|
||||
"id": "init_bq"
|
||||
},
|
||||
"source": [
|
||||
"### Create BigQuery client\n",
|
||||
"\n",
|
||||
"Create the BigQuery client."
|
||||
]
|
||||
},
|
||||
@@ -427,7 +415,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bqclient = bigquery.Client()"
|
||||
"bqclient = bigquery.Client(project=PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -436,17 +424,17 @@
|
||||
"id": "accelerators:prediction,mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"#### Set hardware accelerators\n",
|
||||
"### Set hardware accelerators\n",
|
||||
"\n",
|
||||
"You can set hardware accelerators for prediction.\n",
|
||||
"\n",
|
||||
"Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n",
|
||||
"\n",
|
||||
" (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
|
||||
" (aiplatform.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
|
||||
"\n",
|
||||
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
|
||||
"\n",
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region"
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -457,13 +445,15 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_DEPLOY_GPU\"):\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_DEPLOY_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (aip.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -472,7 +462,7 @@
|
||||
"id": "container:prediction"
|
||||
},
|
||||
"source": [
|
||||
"#### Set pre-built containers\n",
|
||||
"### Set pre-built containers\n",
|
||||
"\n",
|
||||
"Set the pre-built Docker container image for prediction.\n",
|
||||
"\n",
|
||||
@@ -519,11 +509,11 @@
|
||||
"id": "machine:prediction"
|
||||
},
|
||||
"source": [
|
||||
"#### Set machine type\n",
|
||||
"### Set machine type\n",
|
||||
"\n",
|
||||
"Next, set the machine type to use for prediction.\n",
|
||||
"\n",
|
||||
"- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n",
|
||||
"- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM which is used for prediction.\n",
|
||||
" - `machine type`\n",
|
||||
" - `n1-standard`: 3.75GB of memory per vCPU.\n",
|
||||
" - `n1-highmem`: 6.5GB of memory per vCPU\n",
|
||||
@@ -557,7 +547,7 @@
|
||||
"id": "bqml_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Bigquery ML introduction\n",
|
||||
"## BigQuery ML introduction\n",
|
||||
"\n",
|
||||
"BigQuery ML (BQML) provides the capability to train ML tabular models, such as classification and regression, in BigQuery using SQL syntax.\n",
|
||||
"\n",
|
||||
@@ -582,9 +572,9 @@
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"source": [
|
||||
"### Create BQ dataset/model resource\n",
|
||||
"### Create BQ dataset resource\n",
|
||||
"\n",
|
||||
"First, you create a empty dataset/model resource in your project."
|
||||
"First, you create an empty dataset resource in your project."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -659,7 +649,7 @@
|
||||
"id": "bqml_eval_model"
|
||||
},
|
||||
"source": [
|
||||
"### Evaluate the BQML trained model\n",
|
||||
"### Evaluate the trained BQML model\n",
|
||||
"\n",
|
||||
"Next, retrieve the model evaluation for the trained BQML model.\n",
|
||||
"\n",
|
||||
@@ -694,7 +684,7 @@
|
||||
"source": [
|
||||
"### Export the model from BQML\n",
|
||||
"\n",
|
||||
"The model you trained in BQML is a TensorFlow model. Next, you will export the TensorFlow model artifacts in TF.SavedModel format."
|
||||
"The model you trained in BQML is a TensorFlow model. Next, you export the TensorFlow model artifacts in TF.SavedModel format."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -718,9 +708,25 @@
|
||||
"id": "upload_bqml_model"
|
||||
},
|
||||
"source": [
|
||||
"## Upload the BQML model to a Model resource\n",
|
||||
"## Upload the BigQuery ML model to a Vertex AI Model resource\n",
|
||||
"\n",
|
||||
"Finally, now that you have the BQML model exported as a TF.SavedModel format, you upload the model artifacts to Vertex AI Model resource, in the same way as if you were uploading a custom trained model."
|
||||
"Finally, now that you have the BigQuery ML model exported, you upload the model artifacts to Vertex AI Model resource, in the same way as if you were uploading a custom trained model.\n",
|
||||
"\n",
|
||||
"Below is a partial list of mapping BigQuery ML model types to their corresponding exported model format:\n",
|
||||
"\n",
|
||||
"'LINEAR_REG'<br/>\n",
|
||||
"'LOGISTIC_REG' --> TensorFlow SavedFormat\n",
|
||||
"\n",
|
||||
"'AUTOML_CLASSIFIER'<br/>\n",
|
||||
"'AUTOML_REGRESSOR' --> TensorFlow SavedFormat\n",
|
||||
"\n",
|
||||
"'BOOSTED_TREE_CLASSIFIER'<br/>\n",
|
||||
"'BOOSTED_TREE_REGRESSOR' --> XGBoost format\n",
|
||||
"\n",
|
||||
"'DNN_CLASSIFIER'<br/>\n",
|
||||
"'DNN_REGRESSOR'<br/>\n",
|
||||
"'DNN_LINEAR_COMBINED_CLASSIFIER'<br/>\n",
|
||||
"'DNN_LINEAR_COMBINED_REGRESSOR' --> TensorFlow Estimator"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -731,7 +737,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"model = aip.Model.upload(\n",
|
||||
"model = aiplatform.Model.upload(\n",
|
||||
" display_name=\"penguins_\" + TIMESTAMP,\n",
|
||||
" artifact_uri=MODEL_DIR,\n",
|
||||
" serving_container_image_uri=DEPLOY_IMAGE,\n",
|
||||
@@ -752,7 +758,7 @@
|
||||
"- `deployed_model_display_name`: A human readable name for the deployed model.\n",
|
||||
"- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n",
|
||||
"If only one model, then specify as { \"0\": 100 }, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n",
|
||||
"If there are existing models on the endpoint, for which the traffic will be split, then use model_id to specify as { \"0\": percent, model_id: percent, ... }, where model_id is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n",
|
||||
"If there are existing models on the endpoint, for which the traffic needs to be split, then use model_id to specify as { \"0\": percent, model_id: percent, ... }, where model_id is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n",
|
||||
"- `machine_type`: The type of machine to use for training.\n",
|
||||
"- `accelerator_type`: The hardware accelerator type.\n",
|
||||
"- `accelerator_count`: The number of accelerators to attach to a worker replica.\n",
|
||||
@@ -801,7 +807,7 @@
|
||||
"id": "undeploy_model:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"## Undeploy the model\n",
|
||||
"#### Undeploy the model\n",
|
||||
"\n",
|
||||
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
|
||||
]
|
||||
@@ -825,7 +831,7 @@
|
||||
"source": [
|
||||
"#### Delete the model\n",
|
||||
"\n",
|
||||
"The method 'delete()' will delete the model."
|
||||
"The method 'delete()' deletes the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -847,7 +853,7 @@
|
||||
"source": [
|
||||
"### Hyperparameter Tune and train a BQML model\n",
|
||||
"\n",
|
||||
"Next, you train a BQML tabular classification model with hyperparameter tuning using the Vertex AI Vizier service. The hyperparameter settings are specified in the `OPTIONS` statement as follows:\n",
|
||||
"Next, you train a BQML tabular classification model with hyperparameter tuning using the `Vertex AI Vizier` service. The hyperparameter settings are specified in the `OPTIONS` statement as follows:\n",
|
||||
"\n",
|
||||
"- `HPARAM_TUNING_ALGORITHM`: The algorithm for selecting the next trial parameters.\n",
|
||||
"- `num_trials`: The number of trials.\n",
|
||||
@@ -902,7 +908,7 @@
|
||||
"source": [
|
||||
"### Evaluate the BQML trained model\n",
|
||||
"\n",
|
||||
"Next, retrieve the model evaluation for the trained BQML model.\n",
|
||||
"Next, retrieve the model evaluation results for the trained BQML model.\n",
|
||||
"\n",
|
||||
"Learn more about [The ML.EVALUATE function](https://cloud.google.com/bigquery-ml/docs/reference/standard-sql/bigqueryml-syntax-evaluate)."
|
||||
]
|
||||
@@ -976,6 +982,116 @@
|
||||
"print(\"{} created in {}\".format(tblname, job.ended - job.started))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "1bb996026c94"
|
||||
},
|
||||
"source": [
|
||||
"## Model Registry\n",
|
||||
"\n",
|
||||
"Alternatively, you can implicitly upload your BigQuery ML model as a `Vertex AI Model` resource with exporting and importing the model artifacts. In this method, you add additional options when training the model that tells BigQuery ML to automatically upload and register the trained model as a `Model` resource.\n",
|
||||
"\n",
|
||||
"### Setting permissions to automatically register the model\n",
|
||||
"\n",
|
||||
"You need to set some additional IAM permissions for BigQuery ML to automatically upload and register the model after training. Depending on your service account, the setting of the permissions below may fail. In this case, we recommend executing the permissions in a Cloud Shell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "eaaf24146aad"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud projects add-iam-policy-binding $PROJECT_ID \\\n",
|
||||
" --member='serviceAccount:cloud-dataengine@system.gserviceaccount.com' \\\n",
|
||||
" --role='roles/aiplatform.admin'\n",
|
||||
"\n",
|
||||
"! gcloud projects add-iam-policy-binding $PROJECT_ID \\\n",
|
||||
" --member='user:cloud-dataengine@prod.google.com' \\\n",
|
||||
" --role='roles/aiplatform.admin'"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "8bfc0b26155f"
|
||||
},
|
||||
"source": [
|
||||
"### Training and registering the model\n",
|
||||
"\n",
|
||||
"Next, you train the model and automatically register the model to the `Vertex AI Model Registry`, by adding the following parameters as options:\n",
|
||||
"\n",
|
||||
"- `model_registry`: Set to \"vertex_ai\" to indicate automatic registation to `Vertex AI Model Registry`.\n",
|
||||
"- `vertex_ai_model_id`: The human readable display name for the registered model.\n",
|
||||
"- `vertex_ai_model_version_aliases`: Alternate names for the model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "3ff1d4ef4df2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_NAME = \"penguins\"\n",
|
||||
"MODEL_QUERY = f\"\"\"\n",
|
||||
"CREATE OR REPLACE MODEL `{BQ_DATASET_NAME}.{MODEL_NAME}`\n",
|
||||
"OPTIONS(\n",
|
||||
" model_type='DNN_CLASSIFIER',\n",
|
||||
" labels = ['species'],\n",
|
||||
" model_registry=\"vertex_ai\",\n",
|
||||
" vertex_ai_model_id=\"bqml_model_{TIMESTAMP}\", \n",
|
||||
" vertex_ai_model_version_aliases=[\"1\"]\n",
|
||||
" )\n",
|
||||
"AS\n",
|
||||
"SELECT *\n",
|
||||
"FROM `{BQ_TABLE}`\n",
|
||||
"\"\"\"\n",
|
||||
"\n",
|
||||
"job = bqclient.query(MODEL_QUERY)\n",
|
||||
"print(job.errors, job.state)\n",
|
||||
"\n",
|
||||
"while job.running():\n",
|
||||
" from time import sleep\n",
|
||||
"\n",
|
||||
" sleep(30)\n",
|
||||
" print(\"Running ...\")\n",
|
||||
"print(job.errors, job.state)\n",
|
||||
"\n",
|
||||
"tblname = job.ddl_target_table\n",
|
||||
"tblname = \"{}.{}\".format(tblname.dataset_id, tblname.table_id)\n",
|
||||
"print(\"{} created in {}\".format(tblname, job.ended - job.started))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "a243d86f9d22"
|
||||
},
|
||||
"source": [
|
||||
"### Find the model in the `Vertex Model Registry`\n",
|
||||
"\n",
|
||||
"Finally, you can use the `Vertex AI Model` list() method with a filter query to find the automatically registered model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4fd9a143d900"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"models = aiplatform.Model.list(filter=\"display_name=bqml_model_\" + TIMESTAMP)\n",
|
||||
"model = models[0]\n",
|
||||
"\n",
|
||||
"print(model.gca_resource)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -989,14 +1105,9 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Dataset\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -1008,61 +1119,21 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"# Delete the endpoint using the Vertex endpoint object\n",
|
||||
"endpoint.undeploy_all()\n",
|
||||
"endpoint.delete()\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Delete the model using the Vertex model object\n",
|
||||
"try:\n",
|
||||
" model.delete()\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Delete the created GCS bucket\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME\n",
|
||||
" # Delete the created BigQuery dataset\n",
|
||||
" ! bq rm -r -f $PROJECT_ID:$BQ_DATASET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -32,6 +32,11 @@
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex Distributed Training\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/master/notebooks/community/ml_ops/stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
@@ -39,8 +44,9 @@
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
"Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -56,7 +62,7 @@
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Vertex Distributed Training."
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Vertex Distributed Training. Please note: There are incompatibilities between Colab and Docker and the Docker section may not work until resolved by the platform."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -87,10 +93,11 @@
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- `MirroredStrategy`: Train on single VM with multiple GPUs.\n",
|
||||
"- `MirroredStrategy`: Train on a single VM with multiple GPUs.\n",
|
||||
"- `MultiWorkerMirroredStrategy`: Train on multiple VMs with automatic setup of replicas.\n",
|
||||
"- `MultiWorkerMirroredStrategy`: Train on multiple VMs with fine grain control of replicas.\n",
|
||||
"- `ReductionServer`: Train on multiple VMS and sync updates across VMS with Vertex AI Reduction Server"
|
||||
"- `ReductionServer`: Train on multiple VMS and sync updates across VMS with `Vertex AI Reduction Server`.\n",
|
||||
"- `TPUTraining`: Train with multiple Cloud TPUs."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -99,6 +106,15 @@
|
||||
"id": "recommendation:mlops,stage2,vertex,distributed_training"
|
||||
},
|
||||
"source": [
|
||||
"### Costs\n",
|
||||
" \n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"Vertex AI\n",
|
||||
"Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/),\n",
|
||||
" to generate a cost estimate based on your projected usage.\n",
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are best practices for when to use Vertex AI Distributed Training:\n",
|
||||
@@ -125,57 +141,64 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
"id": "XkYpRvOQyVYb"
|
||||
},
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"### Install additional packages\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the latest version of Vertex SDK for Python."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
"id": "xs_Kt8RcyXTC"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "TjOXHg2VyajN"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install {USER_FLAG} --upgrade google-cloud-aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
"id": "oQhwq1iozAxh"
|
||||
},
|
||||
"source": [
|
||||
"### Restart the kernel\n",
|
||||
"\n",
|
||||
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
"After you install the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
"id": "zo3YFZXLzCRJ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Automatically restart kernel after installs\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
@@ -205,6 +228,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
@@ -258,7 +283,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
"id": "qohAA9fJulvP"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -280,7 +305,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
"id": "8NKwwe7aulvQ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -289,6 +314,81 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "poKeKYG8ulvQ"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already\n",
|
||||
"authenticated. Skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "MIpJGzF9ulvQ"
|
||||
},
|
||||
"source": [
|
||||
"**If you are using Colab**, run the cell below and follow the instructions\n",
|
||||
"when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"1. In the Cloud Console, go to the [**Create service account key**\n",
|
||||
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
|
||||
"\n",
|
||||
"2. Click **Create service account**.\n",
|
||||
"\n",
|
||||
"3. In the **Service account name** field, enter a name, and\n",
|
||||
" click **Create**.\n",
|
||||
"\n",
|
||||
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
|
||||
"into the filter box, and select\n",
|
||||
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
|
||||
"local environment.\n",
|
||||
"\n",
|
||||
"6. Enter the path to your service account key as the\n",
|
||||
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Vh6KDXB5ulvQ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebooks, then don't execute this code\n",
|
||||
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -340,7 +440,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
"id": "Moosy2rOulvR"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -360,7 +460,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
"id": "56irx2CvulvS"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -405,7 +505,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk"
|
||||
"id": "wbvYPSTDulvS"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -438,7 +538,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "accelerators:training,prediction,ngpu,mbsdk"
|
||||
"id": "PryARdnoulvT"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -480,7 +580,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "container:training,prediction"
|
||||
"id": "LhhUFw2nulvT"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -548,7 +648,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "machine:training"
|
||||
"id": "vytMaukeulvT"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -610,7 +710,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_custom_pp_training_job:mbsdk"
|
||||
"id": "mhw34XoOulvU"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -659,7 +759,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "examine_training_package"
|
||||
"id": "IAaZpZyyulvU"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -708,7 +808,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "taskpy_contents:mirrored,boston"
|
||||
"id": "zKzddzl6ulvV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -764,6 +864,13 @@
|
||||
" strategy = tf.distribute.MultiWorkerMirroredStrategy()\n",
|
||||
" logging.info(\"Multi-worker Strategy distributed training\")\n",
|
||||
" logging.info('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n",
|
||||
" # Single Machine, multiple TPU devices\n",
|
||||
"elif args.distribute == 'tpu':\n",
|
||||
" cluster_resolver = tf.distribute.cluster_resolver.TPUClusterResolver(tpu=\"local\")\n",
|
||||
" tf.config.experimental_connect_to_cluster(cluster_resolver)\n",
|
||||
" tf.tpu.experimental.initialize_tpu_system(cluster_resolver)\n",
|
||||
" strategy = tf.distribute.TPUStrategy(cluster_resolver)\n",
|
||||
" print(\"All devices: \", tf.config.list_logical_devices('TPU'))\n",
|
||||
"\n",
|
||||
"logging.info('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n",
|
||||
"\n",
|
||||
@@ -822,8 +929,11 @@
|
||||
" else:\n",
|
||||
" task_type, task_id = None, None\n",
|
||||
"\n",
|
||||
" if args.distribute==\"tpu\":\n",
|
||||
" save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n",
|
||||
" model.save(args.model_dir, options=save_locally)\n",
|
||||
" # single, mirrored or primary for multiworker\n",
|
||||
" if _is_chief(task_type, task_id):\n",
|
||||
" elif _is_chief(task_type, task_id):\n",
|
||||
" model.save(args.model_dir)\n",
|
||||
" # non-primary workers for multi-workers\n",
|
||||
" else:\n",
|
||||
@@ -857,7 +967,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "tarball_training_script"
|
||||
"id": "LFUHioqTulvV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -882,7 +992,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_pp_training_job:mirrored"
|
||||
"id": "LnUX0UkvulvV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -917,7 +1027,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "iUWHFpPoulvW"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -939,7 +1049,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "model_delete:mbsdk"
|
||||
"id": "-0gqCUTEulvW"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1024,7 +1134,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_custom_pp_training_job:mbsdk"
|
||||
"id": "aXvPN8P6ulvX"
|
||||
},
|
||||
"source": [
|
||||
"### Create and run custom training job\n",
|
||||
@@ -1050,7 +1160,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_custom_pp_training_job:mbsdk"
|
||||
"id": "kYcFsVSEulvX"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1081,7 +1191,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_pp_training_job:multiworker"
|
||||
"id": "GHRxPU32ulvX"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1108,7 +1218,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "92D_hbuVulvX"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a custom training job\n",
|
||||
@@ -1120,7 +1230,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "CqrfWkB3ulvX"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1172,14 +1282,13 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "write_docker_file:training,multiworker"
|
||||
"id": "pGI2viDAulvY"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile custom/Dockerfile\n",
|
||||
"\n",
|
||||
"FROM gcr.io/deeplearning-platform-release/tf2-gpu.2-5\n",
|
||||
"WORKDIR /root\n",
|
||||
"\n",
|
||||
"WORKDIR /\n",
|
||||
"\n",
|
||||
@@ -1205,7 +1314,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "name_container:training"
|
||||
"id": "7P8cdlFtulvY"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1225,7 +1334,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "build_container:training"
|
||||
"id": "jmw5cakNulvY"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1247,7 +1356,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "test_container:training"
|
||||
"id": "jJGLjU-TulvZ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1269,7 +1378,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "register_container:training"
|
||||
"id": "GAXGjae7ulvZ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1293,7 +1402,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "worker_pool_primary"
|
||||
"id": "CEAnXBzCulvZ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1336,7 +1445,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "worker_pool_training"
|
||||
"id": "6dchPSfNulvZ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1372,7 +1481,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "custom_job:worker_pool"
|
||||
"id": "m2VgmqEOulva"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1396,7 +1505,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_job:multiworker"
|
||||
"id": "hg8vnI_Wulva"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1410,7 +1519,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "WT76Sc-culva"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a custom training job\n",
|
||||
@@ -1422,7 +1531,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "I_IxVfuDulva"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1471,7 +1580,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "custom_job:worker_pool"
|
||||
"id": "L8Av8ATVulvb"
|
||||
},
|
||||
"source": [
|
||||
"### Create CustomJob with worker pool specifications\n",
|
||||
@@ -1487,7 +1596,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "custom_job:worker_pool"
|
||||
"id": "TUWEP1Lmulvb"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1499,7 +1608,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "run_custom_job:multiworker"
|
||||
"id": "_95FH8jeulvb"
|
||||
},
|
||||
"source": [
|
||||
"### Run the CustomJob\n",
|
||||
@@ -1511,7 +1620,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_job:multiworker"
|
||||
"id": "IEbrY05Gulvb"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1525,7 +1634,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "8R2Bnmwmulvb"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a custom training job\n",
|
||||
@@ -1537,7 +1646,234 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
"id": "s1geVE3Lulvb"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "tpu_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Cloud TPU Training\n",
|
||||
"\n",
|
||||
"To further speed up trainig, your organization can utilize Google's Cloud Tensor Processing Units (TPU) pods.\n",
|
||||
"\n",
|
||||
"Cloud TPU is the custom-designed machine learning ASIC that powers Google products like Translate, Photos, Search, Assistant, and Gmail. Cloud TPU is designed to run cutting-edge machine learning models with AI services on Google Cloud. And its custom high-speed network offers over 100 petaflops of performance in a single pod.\n",
|
||||
"\n",
|
||||
"Learn more about [Cloud TPU](https://cloud.google.com/tpu)\n",
|
||||
"\n",
|
||||
"*Note*: TPU VM Training is currently an opt-in feature. Your GCP project must first be added to the feature allowlist. Please email your project information(project id/number) to vertex-ai-tpu-vm-training-support@google.com for the allowlist. You will receive an email as soon as your project is ready."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "docker_write:tpu"
|
||||
},
|
||||
"source": [
|
||||
"### Write Docker file for TPU training\n",
|
||||
"\n",
|
||||
"Currently, there is no pre-built Vertex AI Docker image for training with TPUs. No problems, you can make your own, as follows:\n",
|
||||
"\n",
|
||||
"1. Create a vanilla Python 3 image (e.g., `python3:8`).\n",
|
||||
"2. Get and install the TPU library (`libtpu.so`).\n",
|
||||
"3. Copy in your training package"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "nQVPtknpulvb"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile custom/Dockerfile\n",
|
||||
"FROM python:3.8\n",
|
||||
"\n",
|
||||
"WORKDIR /\n",
|
||||
"\n",
|
||||
"# Copies the trainer code to the docker image.\n",
|
||||
"COPY trainer /trainer\n",
|
||||
"\n",
|
||||
"RUN pip3 install tensorflow-datasets\n",
|
||||
"\n",
|
||||
"# Install TPU Tensorflow and dependencies.\n",
|
||||
"# libtpu.so must be under the '/lib' directory.\n",
|
||||
"RUN wget https://storage.googleapis.com/cloud-tpu-tpuvm-artifacts/libtpu/20210525/libtpu.so -O /lib/libtpu.so\n",
|
||||
"RUN chmod 777 /lib/libtpu.so\n",
|
||||
"\n",
|
||||
"RUN wget https://storage.googleapis.com/cloud-tpu-tpuvm-artifacts/tensorflow/20210525/tf_nightly-2.6.0-cp38-cp38-linux_x86_64.whl\n",
|
||||
"RUN pip3 install tf_nightly-2.6.0-cp38-cp38-linux_x86_64.whl\n",
|
||||
"RUN rm tf_nightly-2.6.0-cp38-cp38-linux_x86_64.whl\n",
|
||||
"# Sets up the entry point to invoke the trainer.\n",
|
||||
"ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "docker_push:tpu"
|
||||
},
|
||||
"source": [
|
||||
"### Build and push the Docker image to the Artifact Registry"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "J_d_zEXUulvc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"TRAIN_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tpu-train:latest\"\n",
|
||||
"\n",
|
||||
"os.chdir(\"custom\")\n",
|
||||
"! docker build --quiet --tag={TRAIN_IMAGE} .\n",
|
||||
"! docker push {TRAIN_IMAGE}\n",
|
||||
"os.chdir(\"..\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "worker_pool_tpu"
|
||||
},
|
||||
"source": [
|
||||
"### TPU worker specification pool\n",
|
||||
"\n",
|
||||
"Next, you create the worker specification pool. For TPUs, you do:\n",
|
||||
"\n",
|
||||
"- Create only one worker pool (Primary).\n",
|
||||
"- Set the machine type to `cloud-tpu`.\n",
|
||||
"- Set the accelerator type to a `TPU`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "d514eU7lulvc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Use TPU Accelerators. Temporarily using numeric codes, until types are added to the SDK\n",
|
||||
"# 6 = TPU_V2\n",
|
||||
"# 7 = TPU_V3\n",
|
||||
"TRAIN_TPU, TRAIN_NTPU = (7, 8)\n",
|
||||
"TRAIN_COMPUTE = \"cloud-tpu\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"if not TRAIN_NTPU or TRAIN_NTPU < 2:\n",
|
||||
" TRAIN_STRATEGY = \"single\"\n",
|
||||
"else:\n",
|
||||
" TRAIN_STRATEGY = \"tpu\"\n",
|
||||
"print(TRAIN_STRATEGY)\n",
|
||||
"\n",
|
||||
"EPOCHS = 20\n",
|
||||
"STEPS = 10000\n",
|
||||
"\n",
|
||||
"TRAINER_ARGS = [\n",
|
||||
" \"--epochs=\" + str(EPOCHS),\n",
|
||||
" \"--steps=\" + str(STEPS),\n",
|
||||
" \"--distribute=\" + TRAIN_STRATEGY,\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"WORKER_POOL_SPECS = [\n",
|
||||
" {\n",
|
||||
" \"container_spec\": {\n",
|
||||
" \"args\": TRAINER_ARGS,\n",
|
||||
" \"image_uri\": TRAIN_IMAGE,\n",
|
||||
" },\n",
|
||||
" \"replica_count\": 1,\n",
|
||||
" \"machine_spec\": {\n",
|
||||
" \"machine_type\": TRAIN_COMPUTE,\n",
|
||||
" \"accelerator_type\": TRAIN_TPU,\n",
|
||||
" \"accelerator_count\": TRAIN_NTPU,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"print(WORKER_POOL_SPECS[0])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "RruSqNfrulvc"
|
||||
},
|
||||
"source": [
|
||||
"### Create CustomJob with worker pool specifications\n",
|
||||
"\n",
|
||||
"Next, you create a `CustomJob` for the multi-worker distributed training job:\n",
|
||||
"\n",
|
||||
"-`display_name`: The display name for the custom job.\n",
|
||||
"\n",
|
||||
"-`worker_pool_specs`: The detailed specifications for each worker pool."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2QvSqbbHulvc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DISPLAY_NAME = \"boston_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"job = aip.CustomJob(display_name=DISPLAY_NAME, worker_pool_specs=WORKER_POOL_SPECS)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "Iw4L3UIfulvd"
|
||||
},
|
||||
"source": [
|
||||
"### Run the CustomJob\n",
|
||||
"\n",
|
||||
"Next, you run the custom job."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "zmqCNS78ulvd"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"try:\n",
|
||||
" job.run(sync=True)\n",
|
||||
"except Exception as e:\n",
|
||||
" # may fail in multi-worker to find startup script\n",
|
||||
" print(e)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "gWZoH9QKulvd"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a custom training job\n",
|
||||
"\n",
|
||||
"After a training job is completed, you can delete the training job with the method `delete()`. Prior to completion, a training job can be canceled with the method `cancel()`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Lt8BJ4iBulvd"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1557,13 +1893,7 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -1571,70 +1901,15 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "cleanup"
|
||||
"id": "U98Wzc01ulvd"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_dataset = True\n",
|
||||
"delete_pipeline = True\n",
|
||||
"delete_model = True\n",
|
||||
"delete_endpoint = True\n",
|
||||
"delete_batchjob = True\n",
|
||||
"delete_customjob = True\n",
|
||||
"delete_hptjob = True\n",
|
||||
"delete_bucket = True\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"# Delete the dataset using the Vertex fully qualified identifier for the dataset\n",
|
||||
"try:\n",
|
||||
" if delete_dataset and \"dataset_id\" in globals():\n",
|
||||
" clients[\"dataset\"].delete_dataset(name=dataset_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n",
|
||||
"try:\n",
|
||||
" if delete_pipeline and \"pipeline_id\" in globals():\n",
|
||||
" clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the model using the Vertex fully qualified identifier for the model\n",
|
||||
"try:\n",
|
||||
" if delete_model and \"model_to_deploy_id\" in globals():\n",
|
||||
" clients[\"model\"].delete_model(name=model_to_deploy_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n",
|
||||
"try:\n",
|
||||
" if delete_endpoint and \"endpoint_id\" in globals():\n",
|
||||
" clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the batch job using the Vertex fully qualified identifier for the batch job\n",
|
||||
"try:\n",
|
||||
" if delete_batchjob and \"batch_job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the custom job using the Vertex fully qualified identifier for the custom job\n",
|
||||
"try:\n",
|
||||
" if delete_customjob and \"job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_custom_job(name=job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n",
|
||||
"try:\n",
|
||||
" if delete_hptjob and \"hpt_job_id\" in globals():\n",
|
||||
" clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n",
|
||||
"except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -29,7 +29,7 @@
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Logging and Vertex Experiments\n",
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Logging and Vertex AI Experiments\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
@@ -38,9 +38,15 @@
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -56,7 +62,7 @@
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Logging and Vertex Experiments."
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Logging and Vertex AI Experiments."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -93,7 +99,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices for logging data when experimenting or formal training a model.\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are some of the best practices for logging data when experimenting or formally training a model.\n",
|
||||
"\n",
|
||||
"#### Python Logging\n",
|
||||
"\n",
|
||||
@@ -105,7 +111,14 @@
|
||||
"\n",
|
||||
"#### Experiments\n",
|
||||
"\n",
|
||||
"Use Vertex AI Experiments in conjunction with logging when doing experiments to compare results for different experiment configurations."
|
||||
"Use Vertex AI Experiments in conjunction with logging when performing experiments to compare results for different experiment configurations.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -116,7 +129,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -127,20 +140,17 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-logging $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -178,6 +188,24 @@
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI, Compute Engine, Cloud Storage and Cloud Logging APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,storage_component,logging).\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
@@ -248,7 +276,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -275,6 +306,63 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f3bd8c0d0469"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex AI\" into the filter box, and select **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e0953a00668e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebook, then don't execute this code\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -284,7 +372,7 @@
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
"### Import libraries"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -295,29 +383,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_logging"
|
||||
},
|
||||
"source": [
|
||||
"#### Import logging\n",
|
||||
"import logging\n",
|
||||
"\n",
|
||||
"Import the logging package into your Python environment."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_logging"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import logging"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -339,7 +407,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION)"
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -356,9 +424,9 @@
|
||||
"- Send log output to console.\n",
|
||||
"- Send log output to a file.\n",
|
||||
"\n",
|
||||
"### Logging Levels\n",
|
||||
"### Logging Levels in Python Logging\n",
|
||||
"\n",
|
||||
"The logging levels in order (from least to highest) are, with each level inclusive of the previous level:\n",
|
||||
"The logging levels in order (from least to highest) and each level inclusive of the previous level are :\n",
|
||||
"\n",
|
||||
"1. Informational\n",
|
||||
"2. Warnings\n",
|
||||
@@ -398,7 +466,7 @@
|
||||
"source": [
|
||||
"### Setting logging level\n",
|
||||
"\n",
|
||||
"To set the logging level, you get the logging handler using `getLogger()`. You can have multiple logging handles. When `getLogger()` is called w/o arguments it gets the default handler, named ROOT. With the handler, you set the logging level with the method 'setLevel()`."
|
||||
"To set the logging level, you get the logging handler using `getLogger()`. You can have multiple logging handles. When `getLogger()` is called without any arguments, it gets the default handler named ROOT. With the handler, you set the logging level with the method `setLevel()`."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -445,7 +513,7 @@
|
||||
"source": [
|
||||
"### Output to a local file\n",
|
||||
"\n",
|
||||
"You can preserve your logging output to a file that is local to where the Python script is running with the method `BasicConfig()`, with the following paraneters:\n",
|
||||
"You can preserve your logging output to a file that is local to where the Python script is running with the method `BasicConfig()`, that takes the following parameters:\n",
|
||||
"\n",
|
||||
"- `filename`: The file path to the local file to write the log output to.\n",
|
||||
"- `level`: Sets the level of logging that is written to the logging file.\n",
|
||||
@@ -482,7 +550,7 @@
|
||||
"- Send log output to storage.\n",
|
||||
"- Retrieve log output from storage.\n",
|
||||
"\n",
|
||||
"### Logging Levels\n",
|
||||
"### Logging Levels in Cloud Logging\n",
|
||||
"\n",
|
||||
"The logging levels in order (from least to highest) are, with each level inclusive of the previous level:\n",
|
||||
"\n",
|
||||
@@ -517,7 +585,7 @@
|
||||
"from google.cloud.logging.handlers import CloudLoggingHandler\n",
|
||||
"\n",
|
||||
"# Connect to the Cloud Logging service\n",
|
||||
"cl_client = google.cloud.logging.Client()\n",
|
||||
"cl_client = google.cloud.logging.Client(project=PROJECT_ID)\n",
|
||||
"handler = CloudLoggingHandler(cl_client, name=\"mylog\")\n",
|
||||
"\n",
|
||||
"# Create a logger instance and logging level\n",
|
||||
@@ -539,7 +607,7 @@
|
||||
"source": [
|
||||
"### Logging output\n",
|
||||
"\n",
|
||||
"To log output at specific levels is identical in method, and method names, as in Python logging, except that you use your instance of the cloud logger in place of logging."
|
||||
"Logging output at specific levels is identical to Python logging with respect to method and method names. The only difference is that you use your instance of the cloud logger in place of logging."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -567,7 +635,7 @@
|
||||
"To get the logged output, you:\n",
|
||||
"\n",
|
||||
"1. Retrieve the log handle to the service.\n",
|
||||
"2. Using the handle call the method `list_entries()`\n",
|
||||
"2. Using the handle, call the method `list_entries()`.\n",
|
||||
"3. Iterate through the entries."
|
||||
]
|
||||
},
|
||||
@@ -594,10 +662,10 @@
|
||||
"source": [
|
||||
"## Logging with Vertex AI Experiments and Vertex AI ML Metadata\n",
|
||||
"\n",
|
||||
"You can log results related to training experiments with `Vertex AI Experiments` and `ML Metadata`:\n",
|
||||
"You can log results related to training experiments with `Vertex AI Experiments` and `ML Metadata` including:\n",
|
||||
"\n",
|
||||
"- Preserve results of an experiment.\n",
|
||||
"- Track multiple runs -- i.e., training runs -- within an experiment.\n",
|
||||
"- Track multiple runs i.e., training runs within an experiment.\n",
|
||||
"- Track parameters (configuration) and metrics (results).\n",
|
||||
"- Retrieve and display the logged output.\n",
|
||||
"\n",
|
||||
@@ -612,14 +680,29 @@
|
||||
"source": [
|
||||
"### Create experiment for tracking training related metadata\n",
|
||||
"\n",
|
||||
"Setup tracking the parameters (configuration) and metrics (results) for each experiment:\n",
|
||||
"Setup tracking for parameters (configuration) and metrics (results) in each experiment:\n",
|
||||
"\n",
|
||||
"- `aip.init()` - Create an experiment instance\n",
|
||||
"- `aip.start_run()` - Track a specific run within the experiment.\n",
|
||||
"- `aiplatform.init()` - Create an experiment instance\n",
|
||||
"- `aiplatform.start_run()` - Track a specific run within the experiment.\n",
|
||||
"\n",
|
||||
"Learn more about [Introduction to Vertex AI ML Metadata](https://cloud.google.com/vertex-ai/docs/ml-metadata/introduction)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "1ed46e349cf2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Specify a name for the experiment\n",
|
||||
"EXPERIMENT_NAME = \"[your-experiment-name]\"\n",
|
||||
"\n",
|
||||
"if EXPERIMENT_NAME == \"[your-experiment-name]\":\n",
|
||||
" EXPERIMENT_NAME = \"example-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -628,9 +711,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPERIMENT_NAME = \"example-\" + TIMESTAMP\n",
|
||||
"aip.init(experiment=EXPERIMENT_NAME)\n",
|
||||
"aip.start_run(\"run-1\")"
|
||||
"# Create experiment\n",
|
||||
"aiplatform.init(experiment=EXPERIMENT_NAME)\n",
|
||||
"aiplatform.start_run(\"run-1\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -641,14 +724,14 @@
|
||||
"source": [
|
||||
"### Log parameters for the experiment\n",
|
||||
"\n",
|
||||
"Typically, an experiment is associated with a specific dataset and model architecture. Within an experiment, you may have multiple training runs, where each run tries a different configuration. As examples:\n",
|
||||
"Typically, an experiment is associated with a specific dataset and a model architecture. Within an experiment, you may have multiple training runs, where each run tries a different configuration. For example:\n",
|
||||
"\n",
|
||||
"- Dataset split\n",
|
||||
"- Dataset sampling and boosting\n",
|
||||
"- Depth and width of layers\n",
|
||||
"- Hyperparameters\n",
|
||||
"\n",
|
||||
"These configuration settings are referred to as parameters, which you store their key/value pair using the method `log_params()`"
|
||||
"These configuration settings are referred to as parameters, which you store as key-value pairs using the method `log_params()`"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -663,7 +746,7 @@
|
||||
"hyperparams[\"epochs\"] = 100\n",
|
||||
"hyperparams[\"batch_size\"] = 32\n",
|
||||
"hyperparams[\"learning_rate\"] = 0.01\n",
|
||||
"aip.log_params(hyperparams)"
|
||||
"aiplatform.log_params(hyperparams)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -674,14 +757,14 @@
|
||||
"source": [
|
||||
"### Log metrics for the experiment\n",
|
||||
"\n",
|
||||
"At the completion, or termination, of a run within an experiment, you can log results that you use to compare runs. As examples:\n",
|
||||
"At the completion or termination of a run within an experiment, you can log results that you use to compare runs. For example:\n",
|
||||
"\n",
|
||||
"- Evaluation metrics\n",
|
||||
"- Hyperparameter search selection\n",
|
||||
"- Time to train the model\n",
|
||||
"- Early stop trigger\n",
|
||||
"\n",
|
||||
"These results settings are referred to as metrics, which you store their key/value pair using the method `log_metrics()`"
|
||||
"These results are referred to as metrics, which you store as key-value pairs using the method `log_metrics()`"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -695,7 +778,7 @@
|
||||
"metrics = {}\n",
|
||||
"metrics[\"test_acc\"] = 98.7\n",
|
||||
"metrics[\"train_acc\"] = 99.3\n",
|
||||
"aip.log_metrics(metrics)"
|
||||
"aiplatform.log_metrics(metrics)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -717,36 +800,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPERIMENT_NAME = \"example\"\n",
|
||||
"\n",
|
||||
"experiment_df = aip.get_experiment_df()\n",
|
||||
"experiment_df = aiplatform.get_experiment_df()\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == EXPERIMENT_NAME]\n",
|
||||
"experiment_df.T"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_experiment"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the experiment\n",
|
||||
"\n",
|
||||
"Next, delete the experiment. You will need to get the context via the metadata to delete it."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_experiment"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"c = aiplatform.metadata._Context(EXPERIMENT_NAME)\n",
|
||||
"c.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -760,15 +818,9 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"### Delete the experiment\n",
|
||||
"\n",
|
||||
"Next, delete the experiment. You will need to get the context via the metadata to delete it."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -779,61 +831,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"c = aiplatform.metadata._Context(EXPERIMENT_NAME)\n",
|
||||
"c.delete()"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -32,6 +32,11 @@
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex Tensorboard\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
@@ -39,8 +44,9 @@
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -81,6 +87,75 @@
|
||||
"- Using Vertex AI TensorBoard with Vertex AI Training."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "b132d4ef86d6"
|
||||
},
|
||||
"source": [
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "94a148f11da5"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your local development environment\n",
|
||||
"\n",
|
||||
"**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n",
|
||||
"all the requirements to run this notebook. You can skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "56cb7f08a9e8"
|
||||
},
|
||||
"source": [
|
||||
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
|
||||
"You need the following:\n",
|
||||
"\n",
|
||||
"* The Google Cloud SDK\n",
|
||||
"* Git\n",
|
||||
"* Python 3\n",
|
||||
"* virtualenv\n",
|
||||
"* Jupyter notebook running in a virtual environment with Python 3\n",
|
||||
"\n",
|
||||
"The Google Cloud guide to [Setting up a Python development\n",
|
||||
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
|
||||
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
|
||||
"for meeting these requirements. The following steps provide a condensed set of\n",
|
||||
"instructions:\n",
|
||||
"\n",
|
||||
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
|
||||
"\n",
|
||||
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
|
||||
"\n",
|
||||
"1. [Install\n",
|
||||
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
|
||||
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
|
||||
"\n",
|
||||
"1. To install Jupyter, run `pip3 install jupyter` on the\n",
|
||||
"command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
|
||||
"\n",
|
||||
"1. Open this notebook in the Jupyter Notebook Dashboard.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -89,7 +164,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices for visualizing your training with TensorBoard.\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following are the best practices for visualizing your training with TensorBoard.\n",
|
||||
"\n",
|
||||
"#### Local TensorBoard\n",
|
||||
"\n",
|
||||
@@ -115,6 +190,25 @@
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "020040f91150"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -123,20 +217,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"! pip3 install -U tensorflow==2.8 $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -168,6 +250,39 @@
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d6a00c14b087"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2721ef0202d9"
|
||||
},
|
||||
"source": [
|
||||
"### Set up your Google Cloud project\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com). {TODO: Update the APIs needed for your tutorial. Edit the API names, and update the link to append the API IDs, separating each one with a comma. For example, container.googleapis.com,cloudbuild.googleapis.com}\n",
|
||||
"\n",
|
||||
"1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
|
||||
"\n",
|
||||
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
|
||||
"Cloud SDK uses the right project for all the commands in this notebook.\n",
|
||||
"\n",
|
||||
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -271,6 +386,81 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2700e693f1b3"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already\n",
|
||||
"authenticated. Skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "885395904904"
|
||||
},
|
||||
"source": [
|
||||
"**If you are using Colab**, run the cell below and follow the instructions\n",
|
||||
"when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"1. In the Cloud Console, go to the [**Create service account key**\n",
|
||||
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
|
||||
"\n",
|
||||
"2. Click **Create service account**.\n",
|
||||
"\n",
|
||||
"3. In the **Service account name** field, enter a name, and\n",
|
||||
" click **Create**.\n",
|
||||
"\n",
|
||||
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
|
||||
"into the filter box, and select\n",
|
||||
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
|
||||
"local environment.\n",
|
||||
"\n",
|
||||
"6. Enter the path to your service account key as the\n",
|
||||
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "eff327d0552b"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebooks, then don't execute this code\n",
|
||||
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -294,7 +484,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -305,8 +495,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -326,7 +516,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -346,7 +536,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -386,7 +576,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -410,7 +600,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
"import google.cloud.aiplatform as aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -454,7 +644,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -484,13 +674,15 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (aip.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
" TRAIN_GPU, TRAIN_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -663,9 +855,9 @@
|
||||
"\n",
|
||||
"You can upload your TensorBoard logs and share with others using `tensorboard dev` command. Once uploaded, a URL is returned to open up the TensorBoard instance in a brower for visualizing.\n",
|
||||
"\n",
|
||||
"*Note:* Your TensorBoard instance is publicly visable.\n",
|
||||
"*Note:* Your TensorBoard instance is publicly visible.\n",
|
||||
"\n",
|
||||
"*Note:* In this example, while running within a notebook, the command will freeze since it is waiting for an interactive yes/no input. You can kill the command with a Ctrl C or kernel interupt.\n",
|
||||
"*Note:* This cell is for demonstration purposes and must be ran in a terminal shell. In this example, while running within a notebook, the command will freeze since it is waiting for an interactive yes/no input. You can kill the command with a Ctrl C or kernel interupt.\n",
|
||||
"\n",
|
||||
"Learn more about [What is TensorBoard.dev](https://tensorboard.dev/)."
|
||||
]
|
||||
@@ -678,7 +870,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! tensorboard dev upload --logdir {LOG_DIR} \\\n",
|
||||
"! tensorboard dev upload --logdir logs \\\n",
|
||||
" --name \"Simple experiment with MNIST\" \\\n",
|
||||
" --description \"Training results\" \\\n",
|
||||
" --one_shot"
|
||||
@@ -706,7 +898,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"TENSORBOARD_DISPLAY_NAME = \"example\"\n",
|
||||
"tensorboard = aip.Tensorboard.create(display_name=TENSORBOARD_DISPLAY_NAME)\n",
|
||||
"tensorboard = aiplatform.Tensorboard.create(display_name=TENSORBOARD_DISPLAY_NAME)\n",
|
||||
"tensorboard_resource_name = tensorboard.gca_resource.name\n",
|
||||
"print(\"TensorBoard resource name:\", tensorboard_resource_name)"
|
||||
]
|
||||
@@ -746,9 +938,9 @@
|
||||
"\n",
|
||||
"url = output[1].split(' ')[-1]\n",
|
||||
"\n",
|
||||
"print(url)\n",
|
||||
"#print(url)\n",
|
||||
"\n",
|
||||
"from IPython.core.display import display, HTML\n",
|
||||
"from IPython.display import display, HTML\n",
|
||||
"display(HTML(\"<a href='\" + url + \"'>click here for TensorBoard instance</a>\"))"
|
||||
]
|
||||
},
|
||||
@@ -953,7 +1145,7 @@
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_example.tar.gz"
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_example.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -985,7 +1177,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job = aip.CustomTrainingJob(\n",
|
||||
"job = aiplatform.CustomTrainingJob(\n",
|
||||
" display_name=\"example_\" + TIMESTAMP,\n",
|
||||
" script_path=\"custom/trainer/task.py\",\n",
|
||||
" container_uri=TRAIN_IMAGE,\n",
|
||||
@@ -1021,7 +1213,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, TIMESTAMP)\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, TIMESTAMP)\n",
|
||||
"\n",
|
||||
"EPOCHS = 20\n",
|
||||
"STEPS = 100\n",
|
||||
@@ -1143,14 +1335,8 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -1162,61 +1348,14 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"# Delete the custom training job\n",
|
||||
"job.delete()\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -623,7 +623,7 @@
|
||||
"In summary:\n",
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Write \"hello world\" to the file."
|
||||
]
|
||||
},
|
||||
@@ -848,7 +848,7 @@
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Get the number of epochs to run from the command line (`--model_dir`).\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Repeat appending \"hello world\" to the file, one per epoch."
|
||||
]
|
||||
},
|
||||
@@ -1045,7 +1045,7 @@
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Get the number of epochs to run from the command line (`--model_dir`).\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Repeat appending \"hello world\" to the file, one per epoch."
|
||||
]
|
||||
},
|
||||
|
||||
@@ -0,0 +1,990 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "copyright"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
"# You may obtain a copy of the License at\n",
|
||||
"#\n",
|
||||
"# https://www.apache.org/licenses/LICENSE-2.0\n",
|
||||
"#\n",
|
||||
"# Unless required by applicable law or agreed to in writing, software\n",
|
||||
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
|
||||
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
|
||||
"# See the License for the specific language governing permissions and\n",
|
||||
"# limitations under the License."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : experimentation: get started with Vertex Training for Scikit-Learn\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_training_sklearn.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_training_sklearn.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
"Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "overview:mlops"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with Vertex Training for Scikit-Learn."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "dataset:custom,newsaggr,tcn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [News Aggregation](https://archive.ics.uci.edu/ml/datasets/News+Aggregator) from [ICS Machine Learning Datasets](https://archive.ics.uci.edu/ml/datasets.php). The trained model predicts the news category of the news article."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "objective:mlops,stage2,get_started_vertex_training_sklearn"
|
||||
},
|
||||
"source": [
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this tutorial, you learn how to use `Vertex AI Training` for training a Scikit-Learn custom model.\n",
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex AI Training`\n",
|
||||
"- `Vertex AI Model` resource\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Training using a Python package.\n",
|
||||
"- Report accuracy when hyperparameter tuning.\n",
|
||||
"- Save the model artifacts to Cloud Storage using GCSFuse.\n",
|
||||
"- Create a `Vertex AI Model` resource."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
},
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"You will not need special packages for this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID:\", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_gcloud_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud config set project $PROJECT_ID"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
},
|
||||
"source": [
|
||||
"#### Region\n",
|
||||
"\n",
|
||||
"You can also change the `REGION` variable, which is used for operations\n",
|
||||
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
|
||||
"\n",
|
||||
"- Americas: `us-central1`\n",
|
||||
"- Europe: `europe-west4`\n",
|
||||
"- Asia Pacific: `asia-east1`\n",
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
},
|
||||
"source": [
|
||||
"#### Timestamp\n",
|
||||
"\n",
|
||||
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bucket:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Create a Cloud Storage bucket\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
},
|
||||
"source": [
|
||||
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
},
|
||||
"source": [
|
||||
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_aip:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "accelerators:training,cpu,prediction,cpu,mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"#### Set hardware accelerators\n",
|
||||
"\n",
|
||||
"You can set hardware accelerators for training and prediction.\n",
|
||||
"\n",
|
||||
"Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n",
|
||||
"\n",
|
||||
" (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
|
||||
"\n",
|
||||
"Learn more about [hardware accelerator support for your region](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators).\n",
|
||||
"\n",
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3. This is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "accelerators:training,cpu,prediction,cpu,mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" TRAIN_GPU, TRAIN_NGPU = (None, None)\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING_DEPLOY_GPU\"):\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (\n",
|
||||
" aip.gapic.AcceleratorType.NVIDIA_TESLA_K80,\n",
|
||||
" int(os.getenv(\"IS_TESTING_DEPLOY_GPU\")),\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" DEPLOY_GPU, DEPLOY_NGPU = (None, None)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "container:training,prediction,scilearn"
|
||||
},
|
||||
"source": [
|
||||
"#### Set pre-built containers\n",
|
||||
"\n",
|
||||
"Set the pre-built Docker container image for training and prediction.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"For the latest list, see [Pre-built containers for training](https://cloud.google.com/ai-platform-unified/docs/training/pre-built-containers).\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/ai-platform-unified/docs/predictions/pre-built-containers)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "container:training,prediction,scilearn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"TRAIN_VERSION = \"scikit-learn-cpu.0-23\"\n",
|
||||
"DEPLOY_VERSION = \"sklearn-cpu.0-23\"\n",
|
||||
"\n",
|
||||
"TRAIN_IMAGE = \"{}-docker.pkg.dev/vertex-ai/training/{}:latest\".format(\n",
|
||||
" REGION.split(\"-\")[0], TRAIN_VERSION\n",
|
||||
")\n",
|
||||
"DEPLOY_IMAGE = \"{}-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(\n",
|
||||
" REGION.split(\"-\")[0], DEPLOY_VERSION\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "machine:training"
|
||||
},
|
||||
"source": [
|
||||
"#### Set machine type\n",
|
||||
"\n",
|
||||
"Next, set the machine type to use for training.\n",
|
||||
"\n",
|
||||
"- Set the variable `TRAIN_COMPUTE` to configure the compute resources for the VMs you will use for for training.\n",
|
||||
" - `machine type`\n",
|
||||
" - `n1-standard`: 3.75GB of memory per vCPU.\n",
|
||||
" - `n1-highmem`: 6.5GB of memory per vCPU\n",
|
||||
" - `n1-highcpu`: 0.9 GB of memory per vCPU\n",
|
||||
" - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n",
|
||||
"\n",
|
||||
"*Note: The following is not supported for training:*\n",
|
||||
"\n",
|
||||
" - `standard`: 2 vCPUs\n",
|
||||
" - `highcpu`: 2, 4 and 8 vCPUs\n",
|
||||
"\n",
|
||||
"*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "machine:training"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n",
|
||||
" MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n",
|
||||
"else:\n",
|
||||
" MACHINE_TYPE = \"n1-standard\"\n",
|
||||
"\n",
|
||||
"VCPU = \"4\"\n",
|
||||
"TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n",
|
||||
"print(\"Train machine type\", TRAIN_COMPUTE)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "sklearn_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Introduction to Scikit-learn training\n",
|
||||
"\n",
|
||||
"Once you have trained a Scikit-learn model, you will want to save it at a Cloud Storage location, so it can subsequently be uploaded to a `Vertex AI Model` resource. The Scikit-learn package does not have support to save the model to a Cloud Storage location. Instead, you will do the following steps to save to a Cloud Storage location.\n",
|
||||
"\n",
|
||||
"1. Save the in-memory model to the local filesystem in pickle format (e.g., model.pkl).\n",
|
||||
"2. Create a Cloud Storage storage client.\n",
|
||||
"3. Upload the pickle file as a blob to the specified Cloud Storage location using the Cloud Storage storage client.\n",
|
||||
"\n",
|
||||
"*Note*: You can do hyperparameter tuning with a Scikit-learn model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "examine_training_package:sklearn"
|
||||
},
|
||||
"source": [
|
||||
"### Examine the training package\n",
|
||||
"\n",
|
||||
"#### Package layout\n",
|
||||
"\n",
|
||||
"Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n",
|
||||
"\n",
|
||||
"- PKG-INFO\n",
|
||||
"- README.md\n",
|
||||
"- setup.cfg\n",
|
||||
"- setup.py\n",
|
||||
"- trainer\n",
|
||||
" - \\_\\_init\\_\\_.py\n",
|
||||
" - task.py\n",
|
||||
"\n",
|
||||
"The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n",
|
||||
"\n",
|
||||
"The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n",
|
||||
"\n",
|
||||
"#### Package Assembly\n",
|
||||
"\n",
|
||||
"In the following cells, you will assemble the training package."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "examine_training_package:sklearn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Make folder for Python training script\n",
|
||||
"! rm -rf custom\n",
|
||||
"! mkdir custom\n",
|
||||
"\n",
|
||||
"# Add package information\n",
|
||||
"! touch custom/README.md\n",
|
||||
"\n",
|
||||
"setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n",
|
||||
"! echo \"$setup_cfg\" > custom/setup.cfg\n",
|
||||
"\n",
|
||||
"setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'wget',\\n\\n 'cloudml-hypertune',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n",
|
||||
"! echo \"$setup_py\" > custom/setup.py\n",
|
||||
"\n",
|
||||
"pkg_info = \"Metadata-Version: 1.0\\n\\nName: News Aggregation text classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n",
|
||||
"! echo \"$pkg_info\" > custom/PKG-INFO\n",
|
||||
"\n",
|
||||
"# Make the training subfolder\n",
|
||||
"! mkdir custom/trainer\n",
|
||||
"! touch custom/trainer/__init__.py"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "taskpy_contents:newsaggr,sklearn"
|
||||
},
|
||||
"source": [
|
||||
"### Create the task script for the Python training package\n",
|
||||
"\n",
|
||||
"Next, you create the `task.py` script for driving the training package. Some noteable steps include:\n",
|
||||
"\n",
|
||||
"- Command-line arguments:\n",
|
||||
" - `model-dir`: The location to save the trained model. When using Vertex AI custom training, the location will be specified in the environment variable: `AIP_MODEL_DIR`,\n",
|
||||
" - `dataset_url`: The location of the dataset to download.\n",
|
||||
" - `alpha`: Hyperparameter\n",
|
||||
"- Data preprocessing (`get_data()`):\n",
|
||||
" - Download the dataset and split into training and test.\n",
|
||||
"- Model architecture (`get_model()`):\n",
|
||||
" - Builds the corresponding model architecture.\n",
|
||||
"- Training (`train_model()`):\n",
|
||||
" - Trains the model\n",
|
||||
"- Evaluation (`evaluate_model()`):\n",
|
||||
" - Evaluates the model.\n",
|
||||
" - If hyperparameter tuning, reports the metric for accuracy.\n",
|
||||
"- Model artifact saving\n",
|
||||
" - Saves the model artifacts and evaluation metrics where the Cloud Storage location specified by `model-dir`.\n",
|
||||
" - *Note*: GCSFuse (`/gcs`) is used to do filesystem operations on Cloud Storage buckets."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "taskpy_contents:newsaggr,sklearn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile custom/trainer/task.py\n",
|
||||
"import argparse\n",
|
||||
"import logging\n",
|
||||
"import os\n",
|
||||
"import pickle\n",
|
||||
"import zipfile\n",
|
||||
"from typing import List, Tuple\n",
|
||||
"\n",
|
||||
"import pandas as pd\n",
|
||||
"import wget\n",
|
||||
"from sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer\n",
|
||||
"from sklearn.model_selection import train_test_split\n",
|
||||
"from sklearn.naive_bayes import MultinomialNB\n",
|
||||
"from sklearn.pipeline import Pipeline\n",
|
||||
"import hypertune\n",
|
||||
"\n",
|
||||
"parser = argparse.ArgumentParser()\n",
|
||||
"parser.add_argument('--model-dir', dest='model_dir',\n",
|
||||
" default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n",
|
||||
"parser.add_argument(\"--dataset-url\", dest=\"dataset_url\",\n",
|
||||
" type=str, help=\"Download url for the training data.\")\n",
|
||||
"parser.add_argument('--alpha', dest='alpha',\n",
|
||||
" default=1.0, type=float,\n",
|
||||
" help='Alpha parameters for MultinomialNB')\n",
|
||||
"args = parser.parse_args()\n",
|
||||
"\n",
|
||||
"logging.getLogger().setLevel(logging.INFO)\n",
|
||||
"\n",
|
||||
"def get_data(url: str, test_size: float = 0.2) -> Tuple[List, List, List, List]:\n",
|
||||
" logging.info(\"Downloading training data from: {}\".format(args.dataset_url))\n",
|
||||
"\n",
|
||||
" zip_filepath = wget.download(url, out=\".\")\n",
|
||||
"\n",
|
||||
" with zipfile.ZipFile(zip_filepath, \"r\") as zf:\n",
|
||||
" zf.extract(path=\".\", member=\"newsCorpora.csv\")\n",
|
||||
"\n",
|
||||
" COLUMN_NAMES = [\"id\", \"title\", \"url\", \"publisher\",\n",
|
||||
" \"category\", \"story\", \"hostname\", \"timestamp\"]\n",
|
||||
"\n",
|
||||
" dataframe = pd.read_csv(\n",
|
||||
" \"newsCorpora.csv\", delimiter=\"\t\", names=COLUMN_NAMES, index_col=0\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train, test = train_test_split(dataframe, test_size=test_size)\n",
|
||||
"\n",
|
||||
" x_train, y_train = train[\"title\"].values, train[\"category\"].values\n",
|
||||
" x_test, y_test = test[\"title\"].values, test[\"category\"].values\n",
|
||||
"\n",
|
||||
" return x_train, y_train, x_test, y_test\n",
|
||||
"\n",
|
||||
"def get_model():\n",
|
||||
" logging.info(\"Build model ...\")\n",
|
||||
" model = Pipeline([\n",
|
||||
" (\"vectorizer\", CountVectorizer()),\n",
|
||||
" (\"tfidf\", TfidfTransformer()),\n",
|
||||
" (\"naivebayes\", MultinomialNB(alpha=args.alpha)),\n",
|
||||
" ])\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
"def train_model(model: Pipeline, X_train: List, y_train: List, X_test: List, y_test: List\n",
|
||||
") -> Pipeline:\n",
|
||||
" logging.info(\"Training started ...\")\n",
|
||||
" model.fit(X_train, y_train)\n",
|
||||
" logging.info(\"Training completed\")\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
"def evaluate_model(model: Pipeline, X_train: List, y_train: List, X_test: List, y_test: List\n",
|
||||
") -> float:\n",
|
||||
" score = model.score(X_test, y_test)\n",
|
||||
" logging.info(f\"Evaluation completed with model score: {score}\")\n",
|
||||
"\n",
|
||||
" # report metric for hyperparameter tuning\n",
|
||||
" hpt = hypertune.HyperTune()\n",
|
||||
" hpt.report_hyperparameter_tuning_metric(\n",
|
||||
" hyperparameter_metric_tag='accuracy',\n",
|
||||
" metric_value=score\n",
|
||||
" )\n",
|
||||
" return score\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def export_model_to_gcs(fitted_pipeline: Pipeline, gcs_uri: str) -> str:\n",
|
||||
" \"\"\"Exports trained pipeline to GCS\n",
|
||||
" Parameters:\n",
|
||||
" fitted_pipeline (sklearn.pipelines.Pipeline): the Pipeline object\n",
|
||||
" with data already fitted (trained pipeline object).\n",
|
||||
" gcs_uri (str): GCS path to store the trained pipeline\n",
|
||||
" i.e gs://example_bucket/training-job.\n",
|
||||
" Returns:\n",
|
||||
" export_path (str): Model GCS location\n",
|
||||
" \"\"\"\n",
|
||||
" # Upload model artifact to Cloud Storage\n",
|
||||
" artifact_filename = 'model.pkl'\n",
|
||||
" storage_path = os.path.join(gcs_uri, artifact_filename)\n",
|
||||
"\n",
|
||||
" # Save model artifact to local filesystem (doesn't persist)\n",
|
||||
" with open(storage_path, 'wb') as model_file:\n",
|
||||
" pickle.dump(fitted_pipeline, model_file)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def export_evaluation_report_to_gcs(report: str, gcs_uri: str) -> None:\n",
|
||||
" \"\"\"\n",
|
||||
" Exports training job report to GCS\n",
|
||||
" Parameters:\n",
|
||||
" report (str): Full report in text to sent to GCS\n",
|
||||
" gcs_uri (str): GCS path to store the report\n",
|
||||
" i.e gs://example_bucket/training-job\n",
|
||||
" \"\"\"\n",
|
||||
"\n",
|
||||
" # Upload model artifact to Cloud Storage\n",
|
||||
" artifact_filename = 'report.txt'\n",
|
||||
" storage_path = os.path.join(gcs_uri, artifact_filename)\n",
|
||||
"\n",
|
||||
" # Save model artifact to local filesystem (doesn't persist)\n",
|
||||
" with open(storage_path, 'w') as report_file:\n",
|
||||
" report_file.write(report)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"logging.info(\"Starting custom training job.\")\n",
|
||||
"\n",
|
||||
"data = get_data(args.dataset_url)\n",
|
||||
"model = get_model()\n",
|
||||
"model = train_model(model, *data)\n",
|
||||
"score = evaluate_model(model, *data)\n",
|
||||
"\n",
|
||||
"# export model to gcs using GCSFuse\n",
|
||||
"logging.info(\"Exporting model artifacts ...\")\n",
|
||||
"gs_prefix = 'gs://'\n",
|
||||
"gcsfuse_prefix = '/gcs/'\n",
|
||||
"if args.model_dir.startswith(gs_prefix):\n",
|
||||
" args.model_dir = args.model_dir.replace(gs_prefix, gcsfuse_prefix)\n",
|
||||
" dirpath = os.path.split(args.model_dir)[0]\n",
|
||||
" if not os.path.isdir(dirpath):\n",
|
||||
" os.makedirs(dirpath)\n",
|
||||
"\n",
|
||||
"export_model_to_gcs(model, args.model_dir)\n",
|
||||
"export_evaluation_report_to_gcs(str(score), args.model_dir)\n",
|
||||
"logging.info(f\"Exported model artifacts to GCS bucket: {args.model_dir}\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "tarball_training_script"
|
||||
},
|
||||
"source": [
|
||||
"#### Store training script on your Cloud Storage bucket\n",
|
||||
"\n",
|
||||
"Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "tarball_training_script"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_newsaggr.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_custom_pp_training_job:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Create and run custom training job\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"To train a custom model, you perform two steps: 1) create a custom training job, and 2) run the job.\n",
|
||||
"\n",
|
||||
"#### Create custom training job\n",
|
||||
"\n",
|
||||
"A custom training job is created with the `CustomTrainingJob` class, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `display_name`: The human readable name for the custom training job.\n",
|
||||
"- `container_uri`: The training container image.\n",
|
||||
"\n",
|
||||
"- `python_package_gcs_uri`: The location of the Python training package as a tarball.\n",
|
||||
"- `python_module_name`: The relative path to the training script in the Python package.\n",
|
||||
"- `model_serving_container_uri`: The container image for deploying the model.\n",
|
||||
"\n",
|
||||
"*Note:* There is no requirements parameter. You specify any requirements in the `setup.py` script in your Python package."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_custom_pp_training_job:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DISPLAY_NAME = \"newsaggr_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"job = aip.CustomPythonPackageTrainingJob(\n",
|
||||
" display_name=DISPLAY_NAME,\n",
|
||||
" python_package_gcs_uri=f\"{BUCKET_URI}/trainer_newsaggr.tar.gz\",\n",
|
||||
" python_module_name=\"trainer.task\",\n",
|
||||
" container_uri=TRAIN_IMAGE,\n",
|
||||
" model_serving_container_image_uri=DEPLOY_IMAGE,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "prepare_custom_cmdargs:newsaggr,sklearn"
|
||||
},
|
||||
"source": [
|
||||
"### Prepare your command-line arguments\n",
|
||||
"\n",
|
||||
"Now define the command-line arguments for your custom training container:\n",
|
||||
"\n",
|
||||
"- `args`: The command-line arguments to pass to the executable that is set as the entry point into the container.\n",
|
||||
" - `--model-dir` : For our demonstrations, we use this command-line argument to specify where to store the model artifacts.\n",
|
||||
" - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n",
|
||||
" - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n",
|
||||
" - `--dataset-url`: The location of the dataset to download.\n",
|
||||
" - `--alpha`: Tunable hyperparameter"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "prepare_custom_cmdargs:newsaggr,sklearn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, TIMESTAMP)\n",
|
||||
"DATASET_URL = \"https://archive.ics.uci.edu/ml/machine-learning-databases/00359/NewsAggregatorDataset.zip\"\n",
|
||||
"\n",
|
||||
"DIRECT = False\n",
|
||||
"\n",
|
||||
"if DIRECT:\n",
|
||||
" CMDARGS = [\n",
|
||||
" \"--alpha=\" + str(0.9),\n",
|
||||
" \"--dataset-url=\" + DATASET_URL,\n",
|
||||
" \"--model_dir=\" + MODEL_DIR,\n",
|
||||
" ]\n",
|
||||
"else:\n",
|
||||
" CMDARGS = [\"--alpha=\" + str(0.9), \"--dataset-url=\" + DATASET_URL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "run_custom_job:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"#### Run the custom training job\n",
|
||||
"\n",
|
||||
"Next, you run the custom job to start the training job by invoking the method `run`, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `model_display_name`: The human readable name for the `Model` resource.\n",
|
||||
"- `args`: The command-line arguments to pass to the training script.\n",
|
||||
"- `replica_count`: The number of compute instances for training (replica_count = 1 is single node training).\n",
|
||||
"- `machine_type`: The machine type for the compute instances.\n",
|
||||
"- `accelerator_type`: The hardware accelerator type.\n",
|
||||
"- `accelerator_count`: The number of accelerators to attach to a worker replica.\n",
|
||||
"- `base_output_dir`: The Cloud Storage location to write the model artifacts to.\n",
|
||||
"- `sync`: Whether to block until completion of the job."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_job:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if TRAIN_GPU:\n",
|
||||
" model = job.run(\n",
|
||||
" model_display_name=\"newsaggr_\" + TIMESTAMP,\n",
|
||||
" args=CMDARGS,\n",
|
||||
" replica_count=1,\n",
|
||||
" machine_type=TRAIN_COMPUTE,\n",
|
||||
" accelerator_type=TRAIN_GPU.name,\n",
|
||||
" accelerator_count=TRAIN_NGPU,\n",
|
||||
" base_output_dir=MODEL_DIR,\n",
|
||||
" sync=False,\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" model = job.run(\n",
|
||||
" model_display_name=\"newsaggr_\" + TIMESTAMP,\n",
|
||||
" args=CMDARGS,\n",
|
||||
" replica_count=1,\n",
|
||||
" machine_type=TRAIN_COMPUTE,\n",
|
||||
" base_output_dir=MODEL_DIR,\n",
|
||||
" sync=False,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"model_path_to_deploy = MODEL_DIR"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "list_job"
|
||||
},
|
||||
"source": [
|
||||
"### List a custom training job"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "list_job"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"_job = job.list(filter=f\"display_name={DISPLAY_NAME}\")\n",
|
||||
"print(_job)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "custom_job_wait:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Wait for completion of custom training job\n",
|
||||
"\n",
|
||||
"Next, wait for the custom training job to complete. Alternatively, one can set the parameter `sync` to `True` in the `run()` method to block until the custom training job is completed."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "custom_job_wait:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"model.wait()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a custom training job\n",
|
||||
"\n",
|
||||
"After a training job is completed, you can delete the training job with the method `delete()`. Prior to completion, a training job can be canceled with the method `cancel()`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cleanup:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"# Cleaning up\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"- Model\n",
|
||||
"- Custom Job (already deleted in previous cell)\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b413063dfdcf"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Delete the model using the Vertex model object\n",
|
||||
"model.delete()\n",
|
||||
"\n",
|
||||
"if os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"name": "get_started_vertex_training_sklearn.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -39,10 +39,11 @@
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_vizier.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_vizier.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" </td> \n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
@@ -139,6 +140,25 @@
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "1fd00fa70a2a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -147,20 +167,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -268,7 +276,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type:\"string\"}\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -318,7 +328,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -329,8 +339,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -350,7 +360,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -370,7 +380,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -415,7 +425,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -808,7 +818,7 @@
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz"
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_boston.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -916,7 +926,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"JOB_NAME = \"custom_job_\" + TIMESTAMP\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, JOB_NAME)\n",
|
||||
"\n",
|
||||
"if not TRAIN_NGPU or TRAIN_NGPU < 2:\n",
|
||||
" TRAIN_STRATEGY = \"single\"\n",
|
||||
@@ -948,7 +958,7 @@
|
||||
" \"disk_spec\": disk_spec,\n",
|
||||
" \"python_package_spec\": {\n",
|
||||
" \"executor_image_uri\": TRAIN_IMAGE,\n",
|
||||
" \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n",
|
||||
" \"package_uris\": [BUCKET_URI + \"/trainer_boston.tar.gz\"],\n",
|
||||
" \"python_module\": \"trainer.task\",\n",
|
||||
" \"args\": CMDARGS,\n",
|
||||
" },\n",
|
||||
@@ -1577,14 +1587,6 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -1596,61 +1598,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -0,0 +1,874 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "VBOfRw7ifk8w"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
"# You may obtain a copy of the License at\n",
|
||||
"#\n",
|
||||
"# https://www.apache.org/licenses/LICENSE-2.0\n",
|
||||
"#\n",
|
||||
"# Unless required by applicable law or agreed to in writing, software\n",
|
||||
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
|
||||
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
|
||||
"# See the License for the specific language governing permissions and\n",
|
||||
"# limitations under the License."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : AutoML Image Classfication Training with Customer Managed Encryption Keys (CMEK)\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_with_cmek_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_with_cmek_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "overview:mlops"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with AutoML training with a customer managed encyrption key CMEK."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "dataset:flowers,icn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public #(GCS) bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "objective:mlops,stage3,get_started_automl_pipeline_components"
|
||||
},
|
||||
"source": [
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this tutorial, you learn how to use a customer managed encryption key (CMEK) for `Vertex AI AutoML` training.\n",
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex AI AutoML`\n",
|
||||
"- Customer managed encryption key.\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Creating a customer managed encryption key.\n",
|
||||
"- Creating an image dataset with CMEK encryption.\n",
|
||||
"- Train an AutoML model with CMEK encryption."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
},
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install the Vertex AI SDK and the KMS package for CMEK encryption."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "sBfZtR4X1Dr_"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-kms $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
},
|
||||
"source": [
|
||||
"### Restart the kernel\n",
|
||||
"\n",
|
||||
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Automatically restart kernel after installs\n",
|
||||
" import IPython\n",
|
||||
"\n",
|
||||
" app = IPython.Application.instance()\n",
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID:\", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_gcloud_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud config set project $PROJECT_ID"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
},
|
||||
"source": [
|
||||
"#### Region\n",
|
||||
"\n",
|
||||
"You can also change the `REGION` variable, which is used for operations\n",
|
||||
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
|
||||
"\n",
|
||||
"- Americas: `us-central1`\n",
|
||||
"- Europe: `europe-west4`\n",
|
||||
"- Asia Pacific: `asia-east1`\n",
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
},
|
||||
"source": [
|
||||
"#### Timestamp\n",
|
||||
"\n",
|
||||
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bucket:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Create a Cloud Storage bucket\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
},
|
||||
"source": [
|
||||
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
},
|
||||
"source": [
|
||||
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_aip:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip\n",
|
||||
"from google.cloud import kms"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_NAME)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "mRk9eoTm6Pyi"
|
||||
},
|
||||
"source": [
|
||||
"## Setting up Customer Managed Encryption Keys\n",
|
||||
"\n",
|
||||
"By default, Google Cloud automatically encrypts data when it is stored in Cloud Storage using encryption keys managed by Google. If you have specific compliance or regulatory requirements related to the keys that protect your data, you can use customer-managed encryption keys (CMEK) for your training jobs.\n",
|
||||
"\n",
|
||||
"### Enable KMS API\n",
|
||||
"\n",
|
||||
"First, you enble the [Cloud Key Management Service (KMS)](https://console.cloud.google.com/flows/enableapi?apiid=cloudkms.googleapis.com)\n",
|
||||
"\n",
|
||||
"Learn more about [Customer managed encryption keys (CMEK)](https://cloud.google.com/vertex-ai/docs/general/cmek)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "RD_Pvrg584X3"
|
||||
},
|
||||
"source": [
|
||||
"### Create a key ring\n",
|
||||
"\n",
|
||||
"After you have enabled the KMS API, you create a key ring and a key. Use the helper function `create_key_ring()` to create a key ring, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `project_id`: Your project ID.\n",
|
||||
"- `location`: Your region.\n",
|
||||
"- `key_ring_id`: The unique identifier for your key ring.\n",
|
||||
"\n",
|
||||
"The helper function calls the KMS client method `create_key_ring()` to create your key ring.\n",
|
||||
"\n",
|
||||
"Learn more about [KMS: Create a key ring](https://cloud.google.com/kms/docs/samples/kms-create-key-ring)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "dxRZzbvQnZC7"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"KEY_RING_ID = \"your_cmek_key_ring_id\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_key_ring(project_id, location, key_ring_id):\n",
|
||||
" \"\"\"\n",
|
||||
" Creates a new key ring in Cloud KMS\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" project_id (string): Google Cloud project ID (e.g. 'my-project').\n",
|
||||
" location (string): Cloud KMS location (e.g. 'us-east1').\n",
|
||||
" id (string): ID of the key ring to create (e.g. 'my-key-ring').\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" KeyRing: Cloud KMS key ring.\n",
|
||||
"\n",
|
||||
" \"\"\"\n",
|
||||
"\n",
|
||||
" # Create the client.\n",
|
||||
" client = kms.KeyManagementServiceClient()\n",
|
||||
"\n",
|
||||
" # Build the parent location name.\n",
|
||||
" location_name = f\"projects/{project_id}/locations/{location}\"\n",
|
||||
"\n",
|
||||
" # Build the key ring.\n",
|
||||
" key_ring = {}\n",
|
||||
"\n",
|
||||
" # Call the API.\n",
|
||||
" created_key_ring = client.create_key_ring(\n",
|
||||
" request={\n",
|
||||
" \"parent\": location_name,\n",
|
||||
" \"key_ring_id\": key_ring_id,\n",
|
||||
" \"key_ring\": key_ring,\n",
|
||||
" }\n",
|
||||
" )\n",
|
||||
" print(\"Created key ring: {}\".format(created_key_ring.name))\n",
|
||||
" return created_key_ring\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"key_ring = create_key_ring(\n",
|
||||
" project_id=PROJECT_ID, location=REGION, key_ring_id=KEY_RING_ID\n",
|
||||
")\n",
|
||||
"print(key_ring)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "gCL1-IfFtWXl"
|
||||
},
|
||||
"source": [
|
||||
"### Create a key\n",
|
||||
"\n",
|
||||
"Next, you create your key. Use the helper function `create_key()` with the following parameters:\n",
|
||||
"\n",
|
||||
"- `project_id`: Your project ID.\n",
|
||||
"- `location`: Your region.\n",
|
||||
"- `key_ring_id`: The unique identifier for your key ring.\n",
|
||||
"- `key_id`: The unique identifier for your key.\n",
|
||||
"\n",
|
||||
"The helper function calls the KMS client method `create_cryto_key()` to create your key.\n",
|
||||
"\n",
|
||||
"Learn more about [](https://cloud.google.com/kms/docs/samples/kms-create-key-symmetric-encrypt-decrypt)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "LXcagdmSnYYW"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"KEY_ID = \"your_cmek_key_id\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_key(project_id, location, key_ring_id, key_id):\n",
|
||||
" \"\"\"\n",
|
||||
" Creates a new symmetric encryption/decryption key in Cloud KMS.\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" project_id (string): Google Cloud project ID (e.g. 'my-project').\n",
|
||||
" location (string): Cloud KMS location (e.g. 'us-east1').\n",
|
||||
" key_ring_id (string): ID of the Cloud KMS key ring (e.g. 'my-key-ring').\n",
|
||||
" key_id (string): ID of the key to create (e.g. 'my-symmetric-key').\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" CryptoKey: Cloud KMS key.\n",
|
||||
"\n",
|
||||
" \"\"\"\n",
|
||||
"\n",
|
||||
" # Create the client.\n",
|
||||
" client = kms.KeyManagementServiceClient()\n",
|
||||
"\n",
|
||||
" # Build the parent key ring name.\n",
|
||||
" key_ring_name = client.key_ring_path(project_id, location, key_ring_id)\n",
|
||||
"\n",
|
||||
" # Build the key.\n",
|
||||
" purpose = kms.CryptoKey.CryptoKeyPurpose.ENCRYPT_DECRYPT\n",
|
||||
" algorithm = (\n",
|
||||
" kms.CryptoKeyVersion.CryptoKeyVersionAlgorithm.GOOGLE_SYMMETRIC_ENCRYPTION\n",
|
||||
" )\n",
|
||||
" key = {\n",
|
||||
" \"purpose\": purpose,\n",
|
||||
" \"version_template\": {\n",
|
||||
" \"algorithm\": algorithm,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Call the API.\n",
|
||||
" created_key = client.create_crypto_key(\n",
|
||||
" request={\"parent\": key_ring_name, \"crypto_key_id\": key_id, \"crypto_key\": key}\n",
|
||||
" )\n",
|
||||
" print(\"Created symmetric key: {}\".format(created_key.name))\n",
|
||||
" return created_key\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"key_id = create_key(\n",
|
||||
" project_id=PROJECT_ID, location=REGION, key_ring_id=KEY_RING_ID, key_id=KEY_ID\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(key_id)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3gKDBOqC8Gl5"
|
||||
},
|
||||
"source": [
|
||||
"### Set service account permissions\n",
|
||||
"\n",
|
||||
"Next, you set permissions for your Vertex AI service account to encrypt and decrypt resources using your key.\n",
|
||||
"\n",
|
||||
"Learn more about [Grant Vertex AI permissions](https://cloud.google.com/vertex-ai/docs/general/cmek#grant_permissions)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "6QrRg08Vqfru"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Reference: https://cloud.google.com/vertex-ai/docs/general/cmek#granting_permissions\n",
|
||||
"# Get the service account\n",
|
||||
"SERVICE_ACCOUNT = ! gcloud projects get-iam-policy {PROJECT_ID} \\\n",
|
||||
" --flatten=\"bindings[].members\" \\\n",
|
||||
" --format=\"table(bindings.members)\" \\\n",
|
||||
" --filter=\"bindings.role:roles/aiplatform.serviceAgent\" \\\n",
|
||||
" | grep -oP \"service-.+?@gcp-sa-aiplatform.iam.gserviceaccount.com\"\n",
|
||||
"SERVICE_ACCOUNT = SERVICE_ACCOUNT[0]\n",
|
||||
"\n",
|
||||
"print(f\"Service account is: {SERVICE_ACCOUNT}\")\n",
|
||||
"\n",
|
||||
"# Give permissions\n",
|
||||
"! gcloud kms keys add-iam-policy-binding {KEY_ID} \\\n",
|
||||
" --keyring={KEY_RING_ID} \\\n",
|
||||
" --location={REGION} \\\n",
|
||||
" --project={PROJECT_ID} \\\n",
|
||||
" --member=serviceAccount:{SERVICE_ACCOUNT} \\\n",
|
||||
" --role=roles/cloudkms.cryptoKeyEncrypterDecrypter"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "1e8cd37e5f99"
|
||||
},
|
||||
"source": [
|
||||
"Create the full resource identifier for the created key"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ebAHZg2vlhXL"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ENCRYPTION_SPEC_KEY_NAME = key_id.name"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "Aa_8wrqSkamz"
|
||||
},
|
||||
"source": [
|
||||
"## Initialize Vertex SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the *client* for Vertex AI\n",
|
||||
"\n",
|
||||
"All resources created during this Notebook run will encrypted with the encryption key created above.\n",
|
||||
"\n",
|
||||
"You can override the encryption key at each function call."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project, bucket, and corresponding encryption key.\n",
|
||||
"\n",
|
||||
"All resources created during this session are encrypted with the encryption key you created.\n",
|
||||
"\n",
|
||||
"*Note:* You can override the encryption key at each function call."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ohdgOs69kGNU"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" staging_bucket=BUCKET_NAME,\n",
|
||||
" location=REGION,\n",
|
||||
" encryption_spec_key_name=ENCRYPTION_SPEC_KEY_NAME,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_file:u_dataset,csv"
|
||||
},
|
||||
"source": [
|
||||
"#### Location of Cloud Storage training data.\n",
|
||||
"\n",
|
||||
"Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:flowers,csv,icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = (\n",
|
||||
" \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "35QVNhACqcTJ"
|
||||
},
|
||||
"source": [
|
||||
"# Create `Vertex AI ImageDataset` resource\n",
|
||||
"\n",
|
||||
"Next, you create an `ImageDataset` resource, which will be encrypted using your encryption key."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4OfCqaYRqcTJ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.ImageDataset.create(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "6-bBqipfqcTS"
|
||||
},
|
||||
"source": [
|
||||
"# Launch a Training Job to Create a Model\n",
|
||||
"\n",
|
||||
"Train an AutoML Image Classification model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "aA41rT_mb-rV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job = aiplatform.AutoMLImageTrainingJob(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" multi_label=False,\n",
|
||||
" model_type=\"CLOUD\",\n",
|
||||
" base_model=None,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# This will take around half an hour to run\n",
|
||||
"model = job.run(\n",
|
||||
" dataset=ds,\n",
|
||||
" model_display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.6,\n",
|
||||
" validation_fraction_split=0.2,\n",
|
||||
" test_fraction_split=0.2,\n",
|
||||
" budget_milli_node_hours=8000,\n",
|
||||
" disable_early_stopping=False,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5vhDsMJNqcTW"
|
||||
},
|
||||
"source": [
|
||||
"# Deploy Your Model\n",
|
||||
"\n",
|
||||
"Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Y9GH72wWqcTX"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint = model.deploy()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "nIw1ifPuqcTb"
|
||||
},
|
||||
"source": [
|
||||
"# Predict on Endpoint\n",
|
||||
"- Take one sample from the data imported to the dataset\n",
|
||||
"- This sample will be encoded to base64 and passed to the endpoint for prediction"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H23ISHdHVIZM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_item = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
"\n",
|
||||
"print(test_item, test_label)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "TF_N0kqZU768"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import base64\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances_list)\n",
|
||||
"\n",
|
||||
"print(prediction)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "nWA3qocXfk82"
|
||||
},
|
||||
"source": [
|
||||
"# Undeploy Model from Endpoint"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "V1brMaO_fk82"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint.undeploy_all()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e00750837ca8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# missing\n",
|
||||
"endpoint.delete()\n",
|
||||
"model.delete()\n",
|
||||
"dataset.delete()\n",
|
||||
"\n",
|
||||
"! gcloud kms keys versions destroy key-version \\\n",
|
||||
" --key key {KEY_ID} \\\n",
|
||||
" --keyring={KEY_RING_ID} \\\n",
|
||||
" --location={REGION} "
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "aa95b7fff9b5"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud kms keys list --location {REGION} --keyring {KEY_RING_ID}"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "get_started_with_cmek_training.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -103,10 +103,10 @@
|
||||
"- Add a serving function for online/batch prediction to the custom model.\n",
|
||||
"- Test the custom model with the serving function.\n",
|
||||
"- Evaluate the custom model using Vertex AI Batch Prediction\n",
|
||||
"- Wait for AutoML training job to complete.\n",
|
||||
"- Wait for the AutoML training job to complete.\n",
|
||||
"- Evaluate the AutoML model using Vertex AI Batch Prediction with the same evaluation slices as the custom model.\n",
|
||||
"- Set the evaluation results of the AutoML model as the baseline.\n",
|
||||
"- If the evaluation of the custom model is below baseline, continue to experiment with custom model.\n",
|
||||
"- If the evaluation of the custom model is below baseline, continue to experiment with the custom model.\n",
|
||||
"- If the evaluation of the custom model is above baseline, save the model as the first best model."
|
||||
]
|
||||
},
|
||||
@@ -186,7 +186,9 @@
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -436,7 +438,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -35,17 +35,131 @@ The third stage in MLOps is formalization to develop an automated pipeline proce
|
||||
|
||||
[Get Started with Kubeflow pipelines](get_started_with_kubeflow_pipelines.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Building KFP lightweight Python function components.
|
||||
- Assembling and compiling KFP components into a pipeline.
|
||||
- Executing a KFP pipeline using Vertex AI Pipelines.
|
||||
- Loading component and pipeline definitions from a source code repository.
|
||||
- Building sequential, parallel, multiple output components.
|
||||
- Building control flow into pipelines.
|
||||
```
|
||||
|
||||
[Get Started with BQ and TFDV components](get_started_with_bq_tfdv_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Build and execute a pipeline component for creating a Vertex AI Tabular Dataset from a BigQuery table.
|
||||
- Build and execute a pipeline component for generating TFDV statistics and schema from a Vertex AI Tabular Dataset.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Dataflow components](get_started_with_dataflow_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Build an Apache Beam data pipeline.
|
||||
- Encapsulate the Apache Beam data pipeline with a Dataflow component in a Vertex AI pipeline.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Dataproc components](get_started_with_dataproc_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- DataprocPySparkBatchOp for PySpark batch workloads.
|
||||
- DataprocSparkBatchOp for Spark batch workloads.
|
||||
- DataprocSparkSqlBatchOp for running Spark SQL batch workloads.
|
||||
- DataprocSparkRBatchOp for running SparkR batch workloads.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI AutoML components](get_started_with_automl_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Training a Vertex AI AutoML trained model.
|
||||
- Test the serving binary with a batch prediction job.
|
||||
- Deploying a Vertex AI AutoML trained model.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI Custom Training components](get_started_with_custom_training_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Training a Vertex AI custom trained model.
|
||||
- Test the serving binary with a batch prediction job.
|
||||
- Deploying a Vertex AI custom trained model.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI Hyperparameter Tuning components](get_started_with_hpt_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Hyperparameter tune/train a custom model.
|
||||
- Retrieve the tuned hyperparameter values and metrics to optimize.
|
||||
- If the metrics exceed a specified threshold.
|
||||
- Get the location of the model artifacts for the best tuned model.
|
||||
- Upload the model artifacts to a `Vertex AI Model` resource.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with BQML components](get_started_with_bqml_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Training BigQuery ML model.
|
||||
- Evaluating the BigQuery ML model.
|
||||
- Exporting the BigQuery ML model.
|
||||
- Importing the BigQuery ML model to a Vertex AI model.
|
||||
- Deploy the Vertex AI model.
|
||||
- Execute a Vertex AI pipeline.
|
||||
- Make a prediction with the deployed Vertex AI model.
|
||||
```
|
||||
|
||||
[Get Started with rapid prototyping with BQML and AutoML components](get_started_with_rapid_prototyping_bqml_automl.ipynb)
|
||||
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a BigQuery and Vertex AI training dataset.
|
||||
- Training a BigQuery ML and AutoML model.
|
||||
- Extracting evaluation metrics from the BigQueryML and AutoML models.
|
||||
- Selecting the best trained model.
|
||||
- Deploying the best trained model.
|
||||
- Testing the deployed model infrastructure.
|
||||
```
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 3: Formalization](mlops_formalization.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Obtain resources from the experimentation stage.
|
||||
- Baseline model.
|
||||
- Dataset schema/statistics for baseline model.
|
||||
- Formalize a data preprocessing pipeline.
|
||||
- Extract columns/rows from BigQuery table to local BigQuery table.
|
||||
- Use Tensorflow Data Validation library to determine statistics, schema, and features.
|
||||
- Use Dataflow to preprocess the data.
|
||||
- Create a Vertex AI Dataset.
|
||||
- Formalize a build model architecture pipeline.
|
||||
- Create the Vertex AI Model base model.
|
||||
- Formalize a training pipeline.
|
||||
```
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -38,9 +38,15 @@
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\\\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -67,7 +73,7 @@
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip."
|
||||
"The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower in the given image from the five classes of flowers: daisy, dandelion, rose, sunflower, or tulip."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -86,11 +92,23 @@
|
||||
"- `Vertex AI AutoML`\n",
|
||||
"- `Google Cloud Pipeline Components`\n",
|
||||
"- `Vertex AI Dataset, Model and Endpoint` resources\n",
|
||||
"- `Vertex AI Prediction`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Construct a pipeline for training and deploying a Vertex AI AutoML model.\n",
|
||||
"- Execute a Vertex AI pipeline."
|
||||
"- Construct a pipeline for:\n",
|
||||
" - Training a Vertex AI AutoML trained model.\n",
|
||||
" - Test the serving binary with a batch prediction job.\n",
|
||||
" - Deploying a Vertex AI AutoML trained model.\n",
|
||||
"- Execute a Vertex AI pipeline.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -101,7 +119,7 @@
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
"Install the following packages for executing this MLOps notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -112,20 +130,22 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\"\n",
|
||||
" \n",
|
||||
"! pip3 install tensorflow-io==0.18 $USER_FLAG -q\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform \\\n",
|
||||
" google-cloud-pipeline-components \\\n",
|
||||
" google-cloud-logging \\\n",
|
||||
" pyarrow \\\n",
|
||||
" kfp $USER_FLAG -q"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -136,7 +156,7 @@
|
||||
"source": [
|
||||
"### Restart the kernel\n",
|
||||
"\n",
|
||||
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
"Once you've installed the additional packages, you need to restart the notebook kernel so that it can find the packages."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -233,7 +253,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -260,6 +283,63 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "c38be665ca50"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
|
||||
"\n",
|
||||
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
|
||||
"\n",
|
||||
"1. **Click Create service account**.\n",
|
||||
"\n",
|
||||
"2. In the **Service account name** field, enter a name, and click **Create**.\n",
|
||||
"\n",
|
||||
"3. In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"4. Click Create. A JSON file that contains your key downloads to your local environment.\n",
|
||||
"\n",
|
||||
"5. Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e0953a00668e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebook, then don't execute this code\n",
|
||||
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -283,7 +363,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -294,8 +375,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -315,7 +396,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -335,7 +416,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -346,7 +427,7 @@
|
||||
"source": [
|
||||
"#### Service Account\n",
|
||||
"\n",
|
||||
"**If you don't know your service account**, try to get your service account using `gcloud` command by executing the second cell below."
|
||||
"You use a service account to create Vertex AI Pipeline jobs. If you do not want to use your project's Compute Engine service account, set `SERVICE_ACCOUNT` to another service account ID."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -370,13 +451,15 @@
|
||||
"source": [
|
||||
"if (\n",
|
||||
" SERVICE_ACCOUNT == \"\"\n",
|
||||
" or SERVICE_ACCOUNT is None\n",
|
||||
" or SERVICE_ACCOUNT == \"[your-service-account]\"\n",
|
||||
" or SERVICE_ACCOUNT is None\n",
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
" shell_output = ! gcloud projects describe $PROJECT_ID | sed -nre 's:.*projectNumber\\: (.*):\\1:p'\n",
|
||||
" SERVICE_ACCOUNT = (\n",
|
||||
" shell_output[0].replace(\"'\", \"\") + \"-compute@developer.gserviceaccount.com\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -387,7 +470,7 @@
|
||||
"source": [
|
||||
"#### Set service account access for Vertex AI Pipelines\n",
|
||||
"\n",
|
||||
"Run the following commands to grant your service account access to read and write pipeline artifacts in the bucket that you created in the previous step -- you only need to run these once per service account."
|
||||
"Run the following commands to grant your service account access to read and write pipeline artifacts in the bucket that you created in the previous step. You only need to run this step once per service account."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -398,9 +481,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_NAME\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
|
||||
"\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_NAME"
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -409,32 +492,7 @@
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_aip:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_tf"
|
||||
},
|
||||
"source": [
|
||||
"#### Import TensorFlow\n",
|
||||
"\n",
|
||||
"Import the TensorFlow package into your Python environment."
|
||||
"### Import libraries"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -445,22 +503,14 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import tensorflow as tf"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_kfp"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import base64\n",
|
||||
"import json\n",
|
||||
"\n",
|
||||
"import google.cloud.aiplatform as aiplatform\n",
|
||||
"import tensorflow as tf\n",
|
||||
"from kfp import dsl\n",
|
||||
"from kfp.v2 import compiler\n",
|
||||
"from kfp.v2.dsl import component"
|
||||
"from kfp.v2.dsl import Artifact, Input, Output, component"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -482,7 +532,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_NAME)"
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -522,7 +572,7 @@
|
||||
"- Takes as input the region and Model artifacts returned from an AutoML training component.\n",
|
||||
"- Create a client interface to the Vertex AI Model service (`metadata[\"resource_name\"]).\n",
|
||||
"- Construct the resource ID for the model from the model artifact parameter.\n",
|
||||
"- Retrieve the model evaluation\n",
|
||||
"- Retrieve the model evaluation.\n",
|
||||
"- Return the model evaluation as a string."
|
||||
]
|
||||
},
|
||||
@@ -534,11 +584,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from kfp.v2.dsl import Artifact, Input, Model\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@component(packages_to_install=[\"google-cloud-aiplatform\"])\n",
|
||||
"def evaluateAutoMLModelOp(model: Input[Artifact], region: str) -> str:\n",
|
||||
"def evaluateAutoMLModelOp(\n",
|
||||
" model: Input[Artifact], region: str, model_evaluation: Output[Artifact]\n",
|
||||
"):\n",
|
||||
" import logging\n",
|
||||
"\n",
|
||||
" import google.cloud.aiplatform.gapic as gapic\n",
|
||||
@@ -551,8 +600,7 @@
|
||||
"\n",
|
||||
" model_evaluations = model_service_client.list_model_evaluations(parent=model_id)\n",
|
||||
" model_evaluation = list(model_evaluations)[0]\n",
|
||||
" logging.info(model_evaluation)\n",
|
||||
" return str(model_evaluation)"
|
||||
" logging.info(model_evaluation)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -568,21 +616,31 @@
|
||||
"1. Use the prebuilt component `ImageDatasetCreateOp` to create a Vertex AI Dataset resource, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The import file for the dataset is passed into the pipeline.\n",
|
||||
" - The component returns the dataset resource as `outputs[\"dataset\"]`\n",
|
||||
" - The component returns the dataset resource as `outputs[\"dataset\"]`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"2. Use the prebuilt component `AutoMLImageTrainingJobRunOp` to train a Vertex AI AutoML Model resource, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The dataset is the output from the `ImageDatasetCreateOp`.\n",
|
||||
" - The component returns the model resource as `outputs[\"model\"]`.\n",
|
||||
"3. Use the prebuilt component `EndpointCreateOp` to create a Vertex AI Endpoint to deploy the trained model to, where:\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"3. Use the prebuild component `ModelBatchPredictOp` to do a test batch prediction, where:\n",
|
||||
" - The model is the output from the `AutoMLTrainingJobRunOp`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"4. Use the prebuilt component `EndpointCreateOp` to create a Vertex AI Endpoint to deploy the trained model to, where:\n",
|
||||
" - Since the component has no dependencies on other components, by default it would be executed in parallel with the model training.\n",
|
||||
" - The `after(training_op)` is added to serialize its execution, so its only executed if the training operation completes successfully.\n",
|
||||
" - The component returns the endpoint resource as `outputs[\"endpoint\"]`.\n",
|
||||
"4. Use the prebuilt component `ModelDeployOp` to deploy the trained AutoML model to, where:\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"5. Use the prebuilt component `ModelDeployOp` to deploy the trained AutoML model where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The model is the output from the `AutoMLTrainingJobRunOp`.\n",
|
||||
" - The endpoint is the output from the `EndpointCreateOp`\n",
|
||||
" - The endpoint is the output from the `EndpointCreateOp`.\n",
|
||||
"\n",
|
||||
"*Note:* Since each component is executed as a graph node in its own execution context, you pass the parameter `project` for each component op, in constrast to doing a `aip.init(project=project)` if this was a Python script calling the SDK methods directly within the same execution context."
|
||||
"*Note:* Since each component is executed as a graph node in its own execution context, you pass the parameter `project` for each component op, in constrast to doing a `aiplatform.init(project=project)` if this was a Python script calling the SDK methods directly within the same execution context."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -593,23 +651,28 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/automl_icn_training\".format(BUCKET_NAME)\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/automl_icn_training\".format(BUCKET_URI)\n",
|
||||
"DEPLOY_COMPUTE = \"n1-standard-4\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(\n",
|
||||
" name=\"automl-icn-training\", description=\"AutoML image classification training\"\n",
|
||||
")\n",
|
||||
"def pipeline(\n",
|
||||
" import_file: str, display_name: str, project: str = PROJECT_ID, region: str = REGION\n",
|
||||
" import_file: str,\n",
|
||||
" batch_files: list,\n",
|
||||
" display_name: str,\n",
|
||||
" bucket: str = PIPELINE_ROOT,\n",
|
||||
" project: str = PROJECT_ID,\n",
|
||||
" region: str = REGION,\n",
|
||||
"):\n",
|
||||
" from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"\n",
|
||||
" dataset_op = gcc_aip.ImageDatasetCreateOp(\n",
|
||||
" project=project,\n",
|
||||
" display_name=display_name,\n",
|
||||
" gcs_source=import_file,\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
" import_schema_uri=aiplatform.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" training_op = gcc_aip.AutoMLImageTrainingJobRunOp(\n",
|
||||
@@ -617,7 +680,6 @@
|
||||
" display_name=display_name,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" model_type=\"CLOUD\",\n",
|
||||
" base_model=None,\n",
|
||||
" dataset=dataset_op.outputs[\"dataset\"],\n",
|
||||
" model_display_name=display_name,\n",
|
||||
" training_fraction_split=0.6,\n",
|
||||
@@ -628,20 +690,132 @@
|
||||
"\n",
|
||||
" eval_op = evaluateAutoMLModelOp(model=training_op.outputs[\"model\"], region=region)\n",
|
||||
"\n",
|
||||
" batch_op = gcc_aip.ModelBatchPredictOp(\n",
|
||||
" project=project,\n",
|
||||
" job_display_name=\"batch_predict_job\",\n",
|
||||
" model=training_op.outputs[\"model\"],\n",
|
||||
" gcs_source_uris=batch_files,\n",
|
||||
" gcs_destination_output_uri_prefix=bucket,\n",
|
||||
" instances_format=\"jsonl\",\n",
|
||||
" predictions_format=\"jsonl\",\n",
|
||||
" model_parameters={},\n",
|
||||
" machine_type=DEPLOY_COMPUTE,\n",
|
||||
" starting_replica_count=1,\n",
|
||||
" max_replica_count=1,\n",
|
||||
" ).after(eval_op)\n",
|
||||
"\n",
|
||||
" endpoint_op = gcc_aip.EndpointCreateOp(\n",
|
||||
" project=project,\n",
|
||||
" location=region,\n",
|
||||
" display_name=display_name,\n",
|
||||
" ).after(eval_op)\n",
|
||||
" ).after(batch_op)\n",
|
||||
"\n",
|
||||
" deploy_op = gcc_aip.ModelDeployOp(\n",
|
||||
" _ = gcc_aip.ModelDeployOp(\n",
|
||||
" model=training_op.outputs[\"model\"],\n",
|
||||
" endpoint=endpoint_op.outputs[\"endpoint\"],\n",
|
||||
" automatic_resources_min_replica_count=1,\n",
|
||||
" automatic_resources_max_replica_count=1,\n",
|
||||
" traffic_split={\"0\": 100},\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "get_test_items:batch_prediction"
|
||||
},
|
||||
"source": [
|
||||
"### Get test item(s)\n",
|
||||
"\n",
|
||||
"In the pipeline, you do a batch prediction on your Vertex model. You will use arbitrary examples from the dataset as test items. Don't be concerned that the examples were likely used while training the model. This step is just to demonstrate how to make a prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "get_test_items:automl,icn,csv"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n2\n",
|
||||
"if len(str(test_items[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
"else:\n",
|
||||
" test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
"\n",
|
||||
"print(test_item_1, test_label_1)\n",
|
||||
"print(test_item_2, test_label_2)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "copy_test_items:batch_prediction"
|
||||
},
|
||||
"source": [
|
||||
"### Copy test item(s)\n",
|
||||
"\n",
|
||||
"For the batch prediction, copy the test items over to your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "copy_test_items:batch_prediction"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"file_1 = test_item_1.split(\"/\")[-1]\n",
|
||||
"file_2 = test_item_2.split(\"/\")[-1]\n",
|
||||
"\n",
|
||||
"! gsutil cp $test_item_1 $BUCKET_URI/$file_1\n",
|
||||
"! gsutil cp $test_item_2 $BUCKET_URI/$file_2\n",
|
||||
"\n",
|
||||
"test_item_1 = BUCKET_URI + \"/\" + file_1\n",
|
||||
"test_item_2 = BUCKET_URI + \"/\" + file_2"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "make_batch_file:automl,image"
|
||||
},
|
||||
"source": [
|
||||
"### Make the batch input file\n",
|
||||
"\n",
|
||||
"Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains key/value pairs:\n",
|
||||
"\n",
|
||||
"- `content`: The Cloud Storage path to the image.\n",
|
||||
"- `mime_type`: The content type. In our example, it is a `jpeg` file.\n",
|
||||
"\n",
|
||||
"For example:\n",
|
||||
"\n",
|
||||
" {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "make_batch_file:automl,image"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"gcs_input_uri = BUCKET_URI + \"/test.jsonl\"\n",
|
||||
"with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n",
|
||||
" data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n",
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
" data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n",
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -653,6 +827,7 @@
|
||||
"Next, you compile the pipeline and then exeute it. The pipeline takes the following parameters, which are passed as the dictionary `parameter_values`:\n",
|
||||
"\n",
|
||||
"- `import_file`: The Cloud Storage path to the dataset index file.\n",
|
||||
"- `batch_files`: A list of Cloud Storage paths to the input batch files.\n",
|
||||
"- `display_name`: The display name for the generated Vertex AI resources.\n",
|
||||
"- `project`: The project ID.\n",
|
||||
"- `region`: The region."
|
||||
@@ -670,12 +845,13 @@
|
||||
" pipeline_func=pipeline, package_path=\"automl_icn_training.json\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"pipeline = aip.PipelineJob(\n",
|
||||
"pipeline = aiplatform.PipelineJob(\n",
|
||||
" display_name=\"automl_icn_training\",\n",
|
||||
" template_path=\"automl_icn_training.json\",\n",
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" \"import_file\": IMPORT_FILE,\n",
|
||||
" \"batch_files\": [gcs_input_uri],\n",
|
||||
" \"display_name\": \"flowers\" + TIMESTAMP,\n",
|
||||
" \"project\": PROJECT_ID,\n",
|
||||
" \"region\": REGION,\n",
|
||||
@@ -739,30 +915,73 @@
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/gcp_resources\"\n",
|
||||
" )\n",
|
||||
" EVAL_METRICS = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/evaluation_metrics\"\n",
|
||||
" )\n",
|
||||
" if tf.io.gfile.exists(EXECUTE_OUTPUT):\n",
|
||||
" ! gsutil cat $EXECUTE_OUTPUT\n",
|
||||
" break\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
" elif tf.io.gfile.exists(GCP_RESOURCES):\n",
|
||||
" ! gsutil cat $GCP_RESOURCES\n",
|
||||
" break\n",
|
||||
" return GCP_RESOURCES\n",
|
||||
" elif tf.io.gfile.exists(EVAL_METRICS):\n",
|
||||
" ! gsutil cat $EVAL_METRICS\n",
|
||||
" return EVAL_METRICS\n",
|
||||
"\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
" return None\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"print(\"imagedataset-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"imagedataset-create\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"automlimagetrainingjob-run\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"automlimagetrainingjob-run\")\n",
|
||||
"print(\"image-dataset-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"image-dataset-create\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"automl-image-training-job\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"automl-image-training-job\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"model_id = output[\"artifacts\"][\"model\"][\"artifacts\"][0][\"metadata\"][\"resourceName\"]\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(model_id)\n",
|
||||
"print(\"endpoint-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"endpoint-create\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"endpoint_id = output[\"artifacts\"][\"endpoint\"][\"artifacts\"][0][\"metadata\"][\n",
|
||||
" \"resourceName\"\n",
|
||||
"]\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(endpoint_id)\n",
|
||||
"print(\"model-deploy\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-deploy\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"evaluateautomlmodelop\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"evaluateautomlmodelop\")"
|
||||
"artifacts = print_pipeline_output(pipeline, \"evaluateautomlmodelop\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-batch-predict\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-batch-predict\")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\n",
|
||||
" output[\"artifacts\"][\"batchpredictionjob\"][\"artifacts\"][0][\"metadata\"][\n",
|
||||
" \"gcsOutputDirectory\"\n",
|
||||
" ]\n",
|
||||
")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"batch_job_id = output[\"artifacts\"][\"batchpredictionjob\"][\"artifacts\"][0][\"metadata\"][\n",
|
||||
" \"resourceName\"\n",
|
||||
"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -790,7 +1009,117 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cleanup:mbsdk"
|
||||
"id": "endpoint_load:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"#### Load an endpoint\n",
|
||||
"\n",
|
||||
"The 'Endpoint' initializer will load an endpoint from an endpoint identifier."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "endpoint_load:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint = aiplatform.Endpoint(endpoint_id)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "make_prediction"
|
||||
},
|
||||
"source": [
|
||||
"## Send a online prediction request\n",
|
||||
"\n",
|
||||
"Send a online prediction request to your deployed model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "get_test_item"
|
||||
},
|
||||
"source": [
|
||||
"### Get test item\n",
|
||||
"\n",
|
||||
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used while training the model. This step is just to demonstrate how to make a prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "get_test_item:automl,icn,csv"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_item = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"if len(str(test_item[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
"else:\n",
|
||||
" test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
"\n",
|
||||
"print(test_item, test_label)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "predict_request:mbsdk,icn"
|
||||
},
|
||||
"source": [
|
||||
"### Make the prediction\n",
|
||||
"\n",
|
||||
"Now that your `Model` resource is deployed to an `Endpoint` resource, you can do online predictions by sending prediction requests to the Endpoint resource.\n",
|
||||
"\n",
|
||||
"#### Request\n",
|
||||
"\n",
|
||||
"Since in this example your test item is in a Cloud Storage bucket, you open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, you encode the bytes into base64 which makes the content safe from modification while transmitting binary data over the network.\n",
|
||||
"\n",
|
||||
"The format of each instance is:\n",
|
||||
"\n",
|
||||
" { 'content': { 'b64': base64_encoded_bytes } }\n",
|
||||
"\n",
|
||||
"Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item.\n",
|
||||
"\n",
|
||||
"#### Response\n",
|
||||
"\n",
|
||||
"The response from the `predict()` call is a Python dictionary with the following entries:\n",
|
||||
"\n",
|
||||
"- `ids`: The internal assigned unique identifiers for each prediction request.\n",
|
||||
"- `displayNames`: The class names for each class label.\n",
|
||||
"- `confidences`: The predicted confidence, between 0 and 1, per class label.\n",
|
||||
"- `deployed_model_id`: The Vertex AI identifier for the deployed Model resource which did the predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "predict_request:mbsdk,icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances)\n",
|
||||
"\n",
|
||||
"print(prediction)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "9d347472d5ba"
|
||||
},
|
||||
"source": [
|
||||
"# Cleaning up\n",
|
||||
@@ -798,17 +1127,40 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial.\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"#### Delete the Vertex AI Model, Endpoint and BatchPredictionJob resources\n",
|
||||
"\n",
|
||||
"Undelpoy and delete the Vertex AI Model, Endpoint and BatchPredictionJob resources."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "baa3e1071f7b"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint.undeploy_all()\n",
|
||||
"endpoint.delete()\n",
|
||||
"\n",
|
||||
"model = aiplatform.Model(model_id)\n",
|
||||
"model.delete()\n",
|
||||
"\n",
|
||||
"batch_job = aiplatform.BatchPredictionJob(batch_job_id)\n",
|
||||
"batch_job.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "a802da1f6fa7"
|
||||
},
|
||||
"source": [
|
||||
"#### Delete the Cloud Storage bucket\n",
|
||||
"\n",
|
||||
"Set `delete_bucket` to *True* to delete the Cloud storage bucket used in this notebook."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -819,61 +1171,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,14 +33,20 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -93,6 +99,28 @@
|
||||
"- Execute a Vertex AI pipeline."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "0c997d8d92ce"
|
||||
},
|
||||
"source": [
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -104,6 +132,25 @@
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "1fd00fa70a2a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -112,20 +159,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"! pip3 install -U tensorflow $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
"! pip3 install --upgrade kfp $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -176,6 +212,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
@@ -233,7 +271,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type:\"string\"}\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -260,6 +300,81 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "648aa9824ac6"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already\n",
|
||||
"authenticated. Skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fc52bba17ee3"
|
||||
},
|
||||
"source": [
|
||||
"**If you are using Colab**, run the cell below and follow the instructions\n",
|
||||
"when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"1. In the Cloud Console, go to the [**Create service account key**\n",
|
||||
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
|
||||
"\n",
|
||||
"2. Click **Create service account**.\n",
|
||||
"\n",
|
||||
"3. In the **Service account name** field, enter a name, and\n",
|
||||
" click **Create**.\n",
|
||||
"\n",
|
||||
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
|
||||
"into the filter box, and select\n",
|
||||
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
|
||||
"local environment.\n",
|
||||
"\n",
|
||||
"6. Enter the path to your service account key as the\n",
|
||||
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "535223fa4b84"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebooks, then don't execute this code\n",
|
||||
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -283,7 +398,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -294,8 +409,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -315,7 +430,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -335,7 +450,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -346,7 +461,9 @@
|
||||
"source": [
|
||||
"#### Service Account\n",
|
||||
"\n",
|
||||
"**If you don't know your service account**, try to get your service account using `gcloud` command by executing the second cell below."
|
||||
"**If you don't know your service account**, try to get your service account using `gcloud` command by executing the second cell below.\n",
|
||||
"\n",
|
||||
"*Note:* The code for automatically finding your service account works on a user-managed Workbench AI noteboook. If you are using a fully-managed notebook or colab, you will need to manually enter your service account."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -375,7 +492,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -398,9 +515,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_NAME\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
|
||||
"\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_NAME"
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -482,7 +599,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -597,7 +714,7 @@
|
||||
" return dataset.column_names\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/dataset_bq\".format(BUCKET_NAME)\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/dataset_bq\".format(BUCKET_URI)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(\n",
|
||||
@@ -610,9 +727,9 @@
|
||||
"):\n",
|
||||
" create_op = create_dataset_bq(bq_table, display_name, project)\n",
|
||||
"\n",
|
||||
" source_op = get_dataset_source(create_op.output)\n",
|
||||
" _ = get_dataset_source(create_op.output)\n",
|
||||
"\n",
|
||||
" column_names_op = get_column_names(create_op.output)\n",
|
||||
" _ = get_column_names(create_op.output)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"compiler.Compiler().compile(pipeline_func=pipeline, package_path=\"dataset_bq.json\")\n",
|
||||
@@ -811,7 +928,7 @@
|
||||
" return (stats_file, schema_file)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/dataset_stats\".format(BUCKET_NAME)\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/dataset_stats\".format(BUCKET_URI)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(\n",
|
||||
@@ -819,7 +936,7 @@
|
||||
")\n",
|
||||
"def pipeline(dataset_id: str, label: str, bucket: str):\n",
|
||||
"\n",
|
||||
" stats_op = statistics(dataset_id, label, bucket)\n",
|
||||
" _ = statistics(dataset_id, label, bucket)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"compiler.Compiler().compile(pipeline_func=pipeline, package_path=\"dataset_stats.json\")\n",
|
||||
@@ -831,7 +948,7 @@
|
||||
" parameter_values={\n",
|
||||
" \"dataset_id\": dataset_id,\n",
|
||||
" \"label\": \"mean_temp\",\n",
|
||||
" \"bucket\": BUCKET_NAME,\n",
|
||||
" \"bucket\": BUCKET_URI,\n",
|
||||
" },\n",
|
||||
")\n",
|
||||
"\n",
|
||||
@@ -901,14 +1018,7 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Vertex AI dataset\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -920,61 +1030,17 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# Create reference to Vertex AI dataset created in pipeline\n",
|
||||
"dataset = aip.TabularDataset(dataset_id)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"# delete Vertex AI dataset\n",
|
||||
"dataset.delete()\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,14 +33,20 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bqml_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td> \n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bqml_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bqml_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bqml_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
@@ -99,6 +105,28 @@
|
||||
"- Make a prediction with the deployed Vertex AI model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "0c997d8d92ce"
|
||||
},
|
||||
"source": [
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -114,24 +142,34 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
"id": "1fd00fa70a2a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "-OxtcyNNJ39g"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install -U tensorflow $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
"! pip3 install --upgrade kfp $USER_FLAG\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -149,7 +187,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
"id": "fIuF_ZjxJ39h"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -182,6 +220,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
@@ -235,11 +275,13 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
"id": "c1Rim3ogJ39j"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type:\"string\"}\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -257,7 +299,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
"id": "hdkr5x2jJ39k"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -266,6 +308,81 @@
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "UG2SHSlTJ39k"
|
||||
},
|
||||
"source": [
|
||||
"### Authenticate your Google Cloud account\n",
|
||||
"\n",
|
||||
"**If you are using Google Cloud Notebooks**, your environment is already\n",
|
||||
"authenticated. Skip this step."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "ETQR4H1HJ39k"
|
||||
},
|
||||
"source": [
|
||||
"**If you are using Colab**, run the cell below and follow the instructions\n",
|
||||
"when prompted to authenticate your account via oAuth.\n",
|
||||
"\n",
|
||||
"**Otherwise**, follow these steps:\n",
|
||||
"\n",
|
||||
"1. In the Cloud Console, go to the [**Create service account key**\n",
|
||||
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
|
||||
"\n",
|
||||
"2. Click **Create service account**.\n",
|
||||
"\n",
|
||||
"3. In the **Service account name** field, enter a name, and\n",
|
||||
" click **Create**.\n",
|
||||
"\n",
|
||||
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
|
||||
"into the filter box, and select\n",
|
||||
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
|
||||
"\n",
|
||||
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
|
||||
"local environment.\n",
|
||||
"\n",
|
||||
"6. Enter the path to your service account key as the\n",
|
||||
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "9M66jv07J39l"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"# If you are running this notebook in Colab, run this cell and follow the\n",
|
||||
"# instructions to authenticate your GCP account. This provides access to your\n",
|
||||
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
|
||||
"# requests.\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# If on Google Cloud Notebooks, then don't execute this code\n",
|
||||
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" if \"google.colab\" in sys.modules:\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"\n",
|
||||
" # If you are running this notebook locally, replace the string below with the\n",
|
||||
" # path to your service account key and run this cell to authenticate your GCP\n",
|
||||
" # account.\n",
|
||||
" elif not os.getenv(\"IS_TESTING\"):\n",
|
||||
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -289,7 +406,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
|
||||
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -300,8 +418,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -317,11 +435,11 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
"id": "V97jQQuiJ39m"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -337,11 +455,11 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
"id": "7PN6kSQtJ39m"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -352,14 +470,16 @@
|
||||
"source": [
|
||||
"#### Service Account\n",
|
||||
"\n",
|
||||
"**If you don't know your service account**, try to get your service account using `gcloud` command by executing the second cell below."
|
||||
"**If you don't know your service account**, try to get your service account using `gcloud` command by executing the second cell below.\n",
|
||||
"\n",
|
||||
"*Note:* The code for automatically finding your service account works on a user-managed Workbench AI noteboook. If you are using a fully-managed notebook, you will need to manually enter your service account."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_service_account"
|
||||
"id": "M4WZi4CDJ39n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -381,7 +501,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -400,13 +520,13 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_service_account:pipelines"
|
||||
"id": "mI3IJONMJ39n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_NAME\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
|
||||
"\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_NAME"
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -443,8 +563,7 @@
|
||||
"import json\n",
|
||||
"\n",
|
||||
"from kfp import dsl\n",
|
||||
"from kfp.v2 import compiler\n",
|
||||
"from kfp.v2.dsl import component"
|
||||
"from kfp.v2 import compiler"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -462,7 +581,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_bq"
|
||||
"id": "r7p4Iv8_J39o"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -484,7 +603,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_tf"
|
||||
"id": "_K5tP8oJJ39p"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -506,11 +625,11 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
"id": "uAnLpS9cJ39p"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_NAME)"
|
||||
"aip.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -528,7 +647,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "init_bq"
|
||||
"id": "I9RloZo9J39p"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -558,7 +677,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "accelerators:prediction,mbsdk"
|
||||
"id": "1-mE_7kXJ39p"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -591,7 +710,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "container:prediction"
|
||||
"id": "AmHM8whxJ39q"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -664,11 +783,11 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_bqml_pipeline:tabular"
|
||||
"id": "LvND7iTpJ39r"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PIPELINE_ROOT = f\"{BUCKET_NAME}/bq_query\"\n",
|
||||
"PIPELINE_ROOT = f\"{BUCKET_URI}/bq_query\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(name=\"bq-hello-world\", pipeline_root=PIPELINE_ROOT)\n",
|
||||
@@ -678,7 +797,7 @@
|
||||
" dataset: str,\n",
|
||||
" model: str,\n",
|
||||
" artifact_uri: str,\n",
|
||||
" min_trials: int,\n",
|
||||
" num_trials: int,\n",
|
||||
" deploy_image: str,\n",
|
||||
" machine_type: str,\n",
|
||||
" min_replica_count: int,\n",
|
||||
@@ -690,68 +809,75 @@
|
||||
" location: str = \"US\",\n",
|
||||
" region: str = \"us-central1\",\n",
|
||||
"):\n",
|
||||
" import google_cloud_pipeline_components.experimental.bigquery as gcc_bq\n",
|
||||
" from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
" from google_cloud_pipeline_components.types import artifact_types\n",
|
||||
" from google_cloud_pipeline_components.v1.bigquery import (\n",
|
||||
" BigqueryCreateModelJobOp, BigqueryEvaluateModelJobOp,\n",
|
||||
" BigqueryExportModelJobOp, BigqueryPredictModelJobOp,\n",
|
||||
" BigqueryQueryJobOp)\n",
|
||||
" from google_cloud_pipeline_components.v1.endpoint import (EndpointCreateOp,\n",
|
||||
" ModelDeployOp)\n",
|
||||
" from google_cloud_pipeline_components.v1.model import ModelUploadOp\n",
|
||||
" from kfp.v2.components import importer_node\n",
|
||||
"\n",
|
||||
" bq_dataset = gcc_bq.BigqueryQueryJobOp(\n",
|
||||
" bq_dataset = BigqueryQueryJobOp(\n",
|
||||
" project=project, location=\"US\", query=f\"CREATE SCHEMA {dataset}\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" bq_model = gcc_bq.BigqueryCreateModelJobOp(\n",
|
||||
" bq_model = BigqueryCreateModelJobOp(\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" query=f\"CREATE OR REPLACE MODEL {dataset}.{model} OPTIONS (model_type='dnn_classifier', labels=['{label}'], min_trials={min_trials}) AS SELECT * FROM `{bq_table}` WHERE body_mass_g IS NOT NULL AND sex IS NOT NULL\",\n",
|
||||
" query=f\"CREATE OR REPLACE MODEL {dataset}.{model} OPTIONS (model_type='dnn_classifier', labels=['{label}'], num_trials={num_trials}) AS SELECT * FROM `{bq_table}` WHERE body_mass_g IS NOT NULL AND sex IS NOT NULL\",\n",
|
||||
" ).after(bq_dataset)\n",
|
||||
"\n",
|
||||
" # bq_eval = gcc_bq.BigqueryEvaluateModelJobOp(\n",
|
||||
" # project=PROJECT_ID,\n",
|
||||
" # location=\"US\",\n",
|
||||
" # model_name=\"bqml_tutorial.penguins_model\",\n",
|
||||
" # ).after(bq_model)\n",
|
||||
"\n",
|
||||
" bq_eval = gcc_bq.BigqueryQueryJobOp(\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" query=f\"SELECT * FROM ML.EVALUATE(MODEL {dataset}.{model}) ORDER BY roc_auc desc LIMIT 1\",\n",
|
||||
" _ = BigqueryEvaluateModelJobOp(\n",
|
||||
" project=PROJECT_ID, location=\"US\", model=bq_model.outputs[\"model\"]\n",
|
||||
" ).after(bq_model)\n",
|
||||
"\n",
|
||||
" bq_predict = gcc_bq.BigqueryPredictModelJobOp(\n",
|
||||
" _ = BigqueryPredictModelJobOp(\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" model_name=f\"{dataset}.{model}\",\n",
|
||||
" model=bq_model.outputs[\"model\"],\n",
|
||||
" table_name=f\"`{bq_table}`\",\n",
|
||||
" # query_statement=f\"SELECT * EXCEPT ({label}) FROM {bq_table} WHERE body_mass_g IS NOT NULL AND sex IS NOT NULL\"\n",
|
||||
" job_configuration_query={\n",
|
||||
" \"destinationTable\": {\n",
|
||||
" \"projectId\": f\"`{project}`\",\n",
|
||||
" \"datasetId\": f\"{dataset}\",\n",
|
||||
" \"projectId\": PROJECT_ID,\n",
|
||||
" \"datasetId\": \"bqml_tutorial\",\n",
|
||||
" \"tableId\": \"results_1\",\n",
|
||||
" }\n",
|
||||
" },\n",
|
||||
" ).after(bq_model)\n",
|
||||
"\n",
|
||||
" bq_export = gcc_bq.BigqueryExportModelJobOp(\n",
|
||||
" bq_export = BigqueryExportModelJobOp(\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" model_name=f\"{project}.{dataset}.{model}\",\n",
|
||||
" model=bq_model.outputs[\"model\"],\n",
|
||||
" model_destination_path=artifact_uri,\n",
|
||||
" ).after(bq_model)\n",
|
||||
"\n",
|
||||
" model_upload = gcc_aip.ModelUploadOp(\n",
|
||||
" display_name=display_name,\n",
|
||||
" import_unmanaged_model_task = importer_node.importer(\n",
|
||||
" artifact_uri=artifact_uri,\n",
|
||||
" serving_container_image_uri=deploy_image,\n",
|
||||
" project=project,\n",
|
||||
" location=region,\n",
|
||||
" artifact_class=artifact_types.UnmanagedContainerModel,\n",
|
||||
" metadata={\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": DEPLOY_IMAGE,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ).after(bq_export)\n",
|
||||
"\n",
|
||||
" endpoint = gcc_aip.EndpointCreateOp(\n",
|
||||
" model_upload = ModelUploadOp(\n",
|
||||
" project=project,\n",
|
||||
" display_name=display_name,\n",
|
||||
" unmanaged_container_model=import_unmanaged_model_task.outputs[\"artifact\"],\n",
|
||||
" ).after(import_unmanaged_model_task)\n",
|
||||
"\n",
|
||||
" endpoint = EndpointCreateOp(\n",
|
||||
" project=project,\n",
|
||||
" location=region,\n",
|
||||
" display_name=display_name,\n",
|
||||
" ).after(model_upload)\n",
|
||||
"\n",
|
||||
" deploy_model = gcc_aip.ModelDeployOp(\n",
|
||||
" _ = ModelDeployOp(\n",
|
||||
" model=model_upload.outputs[\"model\"],\n",
|
||||
" endpoint=endpoint.outputs[\"endpoint\"],\n",
|
||||
" dedicated_resources_min_replica_count=min_replica_count,\n",
|
||||
@@ -778,7 +904,7 @@
|
||||
"- `dataset`: The BigQuery dataset component name.\n",
|
||||
"- `model`: The BigQuery model component name.\n",
|
||||
"- `artifact_uri`: The Cloud Storage location to export the BigQuery model artifacts.\n",
|
||||
"- `min_trials`: If greater than one, will perform hyperparameter tuning for the specified number of trials using the Vertex AI Vizier service.\n",
|
||||
"- `num_trials`: If greater than one, will perform hyperparameter tuning for the specified number of trials using the Vertex AI Vizier service.\n",
|
||||
"- `deploy_image`: The container image for serving predictions.\n",
|
||||
"- `machine_type`: The VM for serving predictions.\n",
|
||||
"- `min_replica_count`/`max_replica_count`: The number of virtual machines for auto-scaling predictions.\n",
|
||||
@@ -793,11 +919,23 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_pipeline:bqml"
|
||||
"id": "l2FMs74-J39r"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_DIR = BUCKET_NAME + \"/bqmodel\"\n",
|
||||
"# If DEPLOY_GPU is None, keeping gpu as no accelerator and accelerator_count as 0\n",
|
||||
"accelerator_count = 0\n",
|
||||
"if DEPLOY_GPU:\n",
|
||||
" gpu = DEPLOY_GPU.name\n",
|
||||
" accelerator_count = 1\n",
|
||||
"else:\n",
|
||||
" gpu = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Unspecified accelerator type, which means no accelerator.\n",
|
||||
" accelerator_count = 0\n",
|
||||
"\n",
|
||||
"print(\"gpu=\", gpu)\n",
|
||||
"print(\"accelerator_count=\", accelerator_count)\n",
|
||||
"\n",
|
||||
"MODEL_DIR = BUCKET_URI + \"/bqmodel\"\n",
|
||||
"\n",
|
||||
"compiler.Compiler().compile(pipeline_func=pipeline, package_path=\"bqml.json\")\n",
|
||||
"\n",
|
||||
@@ -811,14 +949,14 @@
|
||||
" \"dataset\": \"bqml_tutorial\",\n",
|
||||
" \"model\": \"penguins_model\",\n",
|
||||
" \"artifact_uri\": MODEL_DIR,\n",
|
||||
" \"min_trials\": 2,\n",
|
||||
" \"num_trials\": 2,\n",
|
||||
" \"deploy_image\": DEPLOY_IMAGE,\n",
|
||||
" \"display_name\": \"penguins\",\n",
|
||||
" \"machine_type\": \"n1-standard-4\",\n",
|
||||
" \"min_replica_count\": 1,\n",
|
||||
" \"max_replica_count\": 1,\n",
|
||||
" \"accelerator_type\": DEPLOY_GPU.name,\n",
|
||||
" \"accelerator_count\": DEPLOY_NGPU,\n",
|
||||
" \"accelerator_type\": gpu,\n",
|
||||
" \"accelerator_count\": accelerator_count,\n",
|
||||
" \"project\": PROJECT_ID,\n",
|
||||
" \"location\": \"US\",\n",
|
||||
" },\n",
|
||||
@@ -843,7 +981,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "view_pipleline_results:bqml"
|
||||
"id": "2OM8zzJXJ39s"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -880,14 +1018,29 @@
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/gcp_resources\"\n",
|
||||
" )\n",
|
||||
" EVAL_METRICS = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/evaluation_metrics\"\n",
|
||||
" )\n",
|
||||
" if tf.io.gfile.exists(EXECUTE_OUTPUT):\n",
|
||||
" ! gsutil cat $EXECUTE_OUTPUT\n",
|
||||
" break\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
" elif tf.io.gfile.exists(GCP_RESOURCES):\n",
|
||||
" ! gsutil cat $GCP_RESOURCES\n",
|
||||
" break\n",
|
||||
" return GCP_RESOURCES\n",
|
||||
" elif tf.io.gfile.exists(EVAL_METRICS):\n",
|
||||
" ! gsutil cat $EVAL_METRICS\n",
|
||||
" return EVAL_METRICS\n",
|
||||
"\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
" return None\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"print(\"bigquery-query-job\")\n",
|
||||
@@ -896,8 +1049,8 @@
|
||||
"print(\"bigquery-create-model-job\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"bigquery-create-model-job\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"bigquery-query-job-2\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"bigquery-query-job-2\")\n",
|
||||
"print(\"bigquery-evaluate-model-job\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"bigquery-evaluate-model-job\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"bigquery-predict-model-job\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"bigquery-predict-model-job\")\n",
|
||||
@@ -907,6 +1060,9 @@
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-upload\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-upload\")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"model_id = output[\"artifacts\"][\"model\"][\"artifacts\"][0][\"metadata\"][\"resourceName\"]\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"endpoint-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"endpoint-create\")\n",
|
||||
@@ -939,39 +1095,13 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_pipeline"
|
||||
"id": "1UTEiNi9J39s"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"pipeline.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "load_saved_model"
|
||||
},
|
||||
"source": [
|
||||
"## Load the saved model\n",
|
||||
"\n",
|
||||
"Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n",
|
||||
"\n",
|
||||
"To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "load_saved_model:model"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"model = tf.keras.models.load_model(MODEL_DIR)\n",
|
||||
"\n",
|
||||
"model.summary()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -987,7 +1117,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "endpoint_load:mbsdk"
|
||||
"id": "gPEt5GMAJ39s"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1002,22 +1132,49 @@
|
||||
"source": [
|
||||
"#### Make prediction instances\n",
|
||||
"\n",
|
||||
"Next, you prepare a prediction request using a synthetic example. The format for making a prediction request to an export BigQuery ML model is the same as the format for an AutoML model trained on the same data:\n",
|
||||
"Next, you prepare a prediction request using a synthetic example. The format for making a prediction request to an export BigQuery ML model is dependent on the exported model format. In the case where `model_type=DNN_CLASSIFIER`, the exported model format is a TensorFlow estimator format. For this format, you use the `raw_predict()`, with the following request format:\n",
|
||||
"\n",
|
||||
" { 'instances': [ {instance_1}, {instance_2}, ... ])\n",
|
||||
" http_body -> {\n",
|
||||
" 'signature_name' : serving_signature,\n",
|
||||
" 'instances': [ {instance_1}, {instance_2}, ... ]\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" instance -> { 'feature_1': value_1, 'feature_2': value_2, ... }"
|
||||
" instance -> { 'feature_1': value_1, 'feature_2': value_2, ... }\n",
|
||||
"\n",
|
||||
" serving_signature -> \"predict\"\n",
|
||||
"\n",
|
||||
"Below is a partial list of mapping BigQuery ML model types to their corresponding exported model format:\n",
|
||||
"\n",
|
||||
"'LINEAR_REG'<br/>\n",
|
||||
"'LOGISTIC_REG' -> TensorFlow SavedFormat\n",
|
||||
"\n",
|
||||
"'AUTOML_CLASSIFIER'<br/>\n",
|
||||
"'AUTOML_REGRESSOR' -> TensorFlow SavedFormat\n",
|
||||
"\n",
|
||||
"'BOOSTED_TREE_CLASSIFIER'<br/>\n",
|
||||
"'BOOSTED_TREE_REGRESSOR' -> XGBoost format\n",
|
||||
"\n",
|
||||
"'DNN_CLASSIFIER'<br/>\n",
|
||||
"'DNN_REGRESSOR'<br/>\n",
|
||||
"'DNN_LINEAR_COMBINED_CLASSIFIER'<br/>\n",
|
||||
"'DNN_LINEAR_COMBINED_REGRESSOR' -> TensorFlow Estimator"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "make_test_items:bqml,penguins"
|
||||
"id": "sesK_MSdJ39t"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"INSTANCES = {\n",
|
||||
"import json\n",
|
||||
"\n",
|
||||
"from google.api import httpbody_pb2\n",
|
||||
"from google.cloud import aiplatform_v1\n",
|
||||
"\n",
|
||||
"DATA = {\n",
|
||||
" \"signature_name\": \"predict\",\n",
|
||||
" \"instances\": [\n",
|
||||
" {\n",
|
||||
" \"island\": \"DREAM\",\n",
|
||||
@@ -1027,8 +1184,15 @@
|
||||
" \"body_mass_g\": 3475.0,\n",
|
||||
" \"sex\": \"FEMALE\",\n",
|
||||
" }\n",
|
||||
" ]\n",
|
||||
"}"
|
||||
" ],\n",
|
||||
"}\n",
|
||||
"\n",
|
||||
"http_body = httpbody_pb2.HttpBody(\n",
|
||||
" data=json.dumps(DATA).encode(\"utf-8\"),\n",
|
||||
" content_type=\"application/json\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"req = aiplatform_v1.RawPredictRequest(http_body=http_body, endpoint=endpoint_id)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1046,11 +1210,16 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "endpoint_predict:mbsdk"
|
||||
"id": "u5_cgdKQJ39t"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"response = endpoint.predict(INSTANCES)\n",
|
||||
"API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n",
|
||||
"client_options = {\"api_endpoint\": API_ENDPOINT}\n",
|
||||
"\n",
|
||||
"pred_client = aip.gapic.PredictionServiceClient(client_options=client_options)\n",
|
||||
"\n",
|
||||
"response = pred_client.raw_predict(req)\n",
|
||||
"print(response)"
|
||||
]
|
||||
},
|
||||
@@ -1069,15 +1238,41 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete:bqml,penguins"
|
||||
"id": "W0rhdoHmJ39t"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"try:\n",
|
||||
" job = bqclient.delete_model(\"bqml_tutorial.penguins_model\")\n",
|
||||
" job = bqclient.delete_model(f\"{PROJECT_ID}.bqml_tutorial.penguins_model\")\n",
|
||||
"except:\n",
|
||||
" pass\n",
|
||||
"job = bqclient.delete_dataset(\"bqml_tutorial\")"
|
||||
"job = bqclient.delete_dataset(f\"{PROJECT_ID}.bqml_tutorial\", delete_contents=True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "e776f9a3bdc4"
|
||||
},
|
||||
"source": [
|
||||
"#### Delete the Vertex AI Model and Endpoint resources\n",
|
||||
"\n",
|
||||
"Next, undelpoy and delete the Vertex AI Model and Endpoint resources."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "63462e0480f0"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint.undeploy_all()\n",
|
||||
"endpoint.delete()\n",
|
||||
"\n",
|
||||
"model = aip.Model(model_id)\n",
|
||||
"model.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1091,82 +1286,22 @@
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
|
||||
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "cleanup:mbsdk"
|
||||
"id": "ufWUEbnZJ39u"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"# Set this to true only if you'd like to delete your bucket\n",
|
||||
"delete_bucket = False\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -125,7 +125,11 @@
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade python-tabulate $USER_FLAG\n",
|
||||
" ! pip3 install -U opencv-python-headless==4.5.2.52 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -375,7 +379,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -449,9 +453,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components.experimental.dataflow import \\\n",
|
||||
" DataflowPythonJobOp\n",
|
||||
"from google_cloud_pipeline_components.experimental.wait_gcp_resources import \\\n",
|
||||
"from google_cloud_pipeline_components.v1.dataflow import DataflowPythonJobOp\n",
|
||||
"from google_cloud_pipeline_components.v1.wait_gcp_resources import \\\n",
|
||||
" WaitGcpResourcesOp"
|
||||
]
|
||||
},
|
||||
@@ -645,7 +648,8 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile requirements.txt\n",
|
||||
"apache-beam"
|
||||
"apache-beam\n",
|
||||
"future"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -908,7 +912,8 @@
|
||||
"source": [
|
||||
"%%writefile requirements.txt\n",
|
||||
"apache-beam\n",
|
||||
"tensorflow-transform==1.2.0"
|
||||
"tensorflow-transform==1.2.0\n",
|
||||
"future"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -935,6 +940,7 @@
|
||||
"\n",
|
||||
"REQUIRED_PACKAGES = [\n",
|
||||
" 'tensorflow-transform==1.2.0',\n",
|
||||
" 'future'\n",
|
||||
"]\n",
|
||||
"PACKAGE_NAME = 'my_package'\n",
|
||||
"PACKAGE_VERSION = '0.0.1'\n",
|
||||
|
||||