Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ccd89d1384 | ||
|
|
0980f4e1b7 | ||
|
|
e5bbae8023 | ||
|
|
efaa340a98 | ||
|
|
a132f83e30 | ||
|
|
e338d7187b | ||
|
|
78e3cb6170 | ||
|
|
77f81441eb | ||
|
|
3b7ca24a9c | ||
|
|
daa64efd40 | ||
|
|
a37deabe27 | ||
|
|
6009ef0def | ||
|
|
edc644b0d0 | ||
|
|
1297af8baf | ||
|
|
b8b1b6675b | ||
|
|
7d3b7abc44 | ||
|
|
9a572f298e | ||
|
|
b53ca9e678 | ||
|
|
9aceec161a | ||
|
|
98b186ed55 | ||
|
|
f23ee1b5a8 | ||
|
|
c6d779f1fc | ||
|
|
8908b27b08 | ||
|
|
8947c9b116 | ||
|
|
9464caac6e | ||
|
|
98ce91c575 | ||
|
|
07f8feda3d | ||
|
|
a4e0496ff5 | ||
|
|
cf162c02c8 | ||
|
|
831aae94df | ||
|
|
3fd9f28778 | ||
|
|
a656e8e2a8 | ||
|
|
2f02152703 | ||
|
|
9d08f8ce67 | ||
|
|
ec6d508793 | ||
|
|
772e35ea75 | ||
|
|
1d28f886c8 | ||
|
|
d3dc8aeb9a | ||
|
|
a07d762934 | ||
|
|
85ac9e127d | ||
|
|
011c2823ff | ||
|
|
f28a94f03f | ||
|
|
5ecfc80cb9 | ||
|
|
e5e36ba050 | ||
|
|
0ad9116d6a | ||
|
|
0516032443 | ||
|
|
95d211c90f | ||
|
|
12d6a75ef7 | ||
|
|
06153dc373 | ||
|
|
02afa91fc3 | ||
|
|
c8b212789f | ||
|
|
95256d3fcf | ||
|
|
8a6d174c99 | ||
|
|
d36cf7f662 | ||
|
|
1b02a542c8 | ||
|
|
45430bb010 | ||
|
|
be2a139ade | ||
|
|
70d77b24f6 | ||
|
|
50c25d6d7b | ||
|
|
5bcdc0bc64 | ||
|
|
eff0f95b58 | ||
|
|
50b53d31bd | ||
|
|
b5a56852f3 | ||
|
|
b7486e34ad | ||
|
|
3edc5f1425 | ||
|
|
65b4b73cb5 | ||
|
|
186c08e8c3 | ||
|
|
7721aa0def | ||
|
|
4987c60e03 | ||
|
|
cd845f7fdd | ||
|
|
9e84d9e782 | ||
|
|
23c7fcc97f | ||
|
|
e4024efbc7 | ||
|
|
c4d53108af | ||
|
|
be8fe3564d | ||
|
|
1f95775057 | ||
|
|
f2a4dd875e | ||
|
|
67dd2300c8 | ||
|
|
b5391b06b4 | ||
|
|
54f2c71c13 | ||
|
|
6697900126 | ||
|
|
9ff3400b44 | ||
|
|
3ff0726ebf | ||
|
|
c96c939dfe | ||
|
|
719cf280c9 | ||
|
|
4ec6e2df04 | ||
|
|
031a9190c3 | ||
|
|
5204dcf327 | ||
|
|
cad623ef84 | ||
|
|
a59f58f8b6 | ||
|
|
d89c613f5d | ||
|
|
fa265ddb2f | ||
|
|
ec3dd04935 | ||
|
|
0c83e81410 | ||
|
|
7808a843cc | ||
|
|
615d7706af | ||
|
|
f05ca4d06a | ||
|
|
cb4145e2b6 | ||
|
|
6917c9aa7b | ||
|
|
c1d2451656 | ||
|
|
c41ec1fabd | ||
|
|
273c91882e | ||
|
|
ad6b5e5830 | ||
|
|
d441eb9d7a | ||
|
|
139d805c9f | ||
|
|
783347fc8e | ||
|
|
6d722d081d | ||
|
|
33a8c6ca0e | ||
|
|
ae043400f5 | ||
|
|
e41f96b31f | ||
|
|
7220f15158 | ||
|
|
b8d7eaa767 | ||
|
|
a612b463a8 | ||
|
|
493e7e81b5 | ||
|
|
b6cf0dbefd | ||
|
|
5d5c08f9e7 | ||
|
|
bfdfac0f81 | ||
|
|
9c72b7e3e7 | ||
|
|
46519e5c64 | ||
|
|
0ff961e203 | ||
|
|
cbc17c6832 | ||
|
|
93e5b15cba | ||
|
|
081e65d076 | ||
|
|
4ab2cfb713 | ||
|
|
a73335c0af | ||
|
|
0f3e257773 | ||
|
|
57d734d84f | ||
|
|
07f2e9c999 | ||
|
|
401883cae3 | ||
|
|
7bc92e1e3c | ||
|
|
de5f8b0653 | ||
|
|
cfa73ed53b | ||
|
|
67370bb1c7 | ||
|
|
f60593255a | ||
|
|
c6f9b97615 | ||
|
|
6161a394c2 | ||
|
|
b35cd42015 | ||
|
|
5e323993db | ||
|
|
f1623e419e | ||
|
|
a3047fb1bb | ||
|
|
b4d02f486e | ||
|
|
9e24893b9b | ||
|
|
47dec6ecef | ||
|
|
059fea672c | ||
|
|
389e804426 | ||
|
|
087a638c18 | ||
|
|
97757c74ca | ||
|
|
08f3ad583b | ||
|
|
16f01d31d3 | ||
|
|
e0f6c66351 | ||
|
|
3733b28772 | ||
|
|
24f8912134 | ||
|
|
cdc8847f9b | ||
|
|
25bd4b9eb5 | ||
|
|
d476191252 | ||
|
|
bf35d6a07c | ||
|
|
5378d38a05 | ||
|
|
85984f1173 | ||
|
|
92bb40349f | ||
|
|
fcee9bf738 | ||
|
|
cc2f3408ad | ||
|
|
89fb218041 | ||
|
|
4d0c3781e5 | ||
|
|
e2e319ac7f | ||
|
|
63508cad3f | ||
|
|
0b6eb6eed1 | ||
|
|
a41cfdeba5 | ||
|
|
7a6763ca33 | ||
|
|
2a6e19e4f9 | ||
|
|
9f9526b722 | ||
|
|
267684b7c3 | ||
|
|
b74dd3d576 | ||
|
|
278ae48842 | ||
|
|
189ca12627 | ||
|
|
5108f57ba8 | ||
|
|
6bace666e4 | ||
|
|
4c35616d30 | ||
|
|
cadbc733b8 | ||
|
|
ee6088957d | ||
|
|
58118c89f7 | ||
|
|
108112c9a0 | ||
|
|
4f586719e7 | ||
|
|
31e4985480 | ||
|
|
672b8bd262 | ||
|
|
c64c185407 | ||
|
|
19fd6aa4e3 | ||
|
|
55746d05f3 | ||
|
|
ad8d6382d4 | ||
|
|
5114a6c09c | ||
|
|
f405c109e2 | ||
|
|
ff646fee07 | ||
|
|
e517a8998c | ||
|
|
776a699a76 | ||
|
|
a942a63959 | ||
|
|
4b275e3370 | ||
|
|
7bb6787800 | ||
|
|
d314dda749 |
@@ -1,194 +0,0 @@
|
||||
# Copyright 2020 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# We want to use LTS ubuntu from our mirror because dockerhub has a
|
||||
# rate limit.
|
||||
# FROM mirror.gcr.io/library/ubuntu:18.04
|
||||
# However, now the above image is not working, we're using our own cache
|
||||
FROM gcr.io/cloud-devrel-kokoro-resources/ubuntu:20.04
|
||||
|
||||
ENV DEBIAN_FRONTEND noninteractive
|
||||
|
||||
# Ensure local Python is preferred over distribution Python.
|
||||
ENV PATH /usr/local/bin:$PATH
|
||||
|
||||
# http://bugs.python.org/issue19846
|
||||
# At the moment, setting "LANG=C" on a Linux system fundamentally breaks
|
||||
# Python 3.
|
||||
ENV LANG C.UTF-8
|
||||
|
||||
# Install dependencies.
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
apt-transport-https \
|
||||
build-essential \
|
||||
ca-certificates \
|
||||
curl \
|
||||
dirmngr \
|
||||
git \
|
||||
gcc \
|
||||
gpg-agent \
|
||||
graphviz \
|
||||
libbz2-dev \
|
||||
libdb5.3-dev \
|
||||
libexpat1-dev \
|
||||
libffi-dev \
|
||||
liblzma-dev \
|
||||
libmagickwand-dev \
|
||||
libmemcached-dev \
|
||||
libpython3-dev \
|
||||
libreadline-dev \
|
||||
libsnappy-dev \
|
||||
libssl-dev \
|
||||
libsqlite3-dev \
|
||||
portaudio19-dev \
|
||||
pkg-config \
|
||||
redis-server \
|
||||
software-properties-common \
|
||||
ssh \
|
||||
sudo \
|
||||
systemd \
|
||||
tcl \
|
||||
tcl-dev \
|
||||
tk \
|
||||
tk-dev \
|
||||
uuid-dev \
|
||||
wget \
|
||||
zlib1g-dev \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
# Install docker
|
||||
RUN curl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo apt-key add -
|
||||
|
||||
RUN add-apt-repository \
|
||||
"deb [arch=amd64] https://download.docker.com/linux/ubuntu \
|
||||
$(lsb_release -cs) \
|
||||
stable"
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
docker-ce \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
# Install Bazel for compiling Tink in Cloud SQL Client Side Encryption Samples
|
||||
# TODO: Delete this section once google/tink#483 is resolved
|
||||
RUN apt install -y curl gpgconf gpg \
|
||||
&& curl -fsSL https://bazel.build/bazel-release.pub.gpg | gpg --dearmor > bazel.gpg \
|
||||
&& mv bazel.gpg /etc/apt/trusted.gpg.d/ \
|
||||
&& echo "deb [arch=amd64] https://storage.googleapis.com/bazel-apt stable jdk1.8" | sudo tee /etc/apt/sources.list.d/bazel.list \
|
||||
&& apt update && apt install -y bazel \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
# Install Microsoft ODBC 17 Driver and unixodbc for testing SQL Server samples
|
||||
RUN curl https://packages.microsoft.com/keys/microsoft.asc | apt-key add - \
|
||||
&& curl https://packages.microsoft.com/config/ubuntu/20.04/prod.list > /etc/apt/sources.list.d/mssql-release.list \
|
||||
&& apt-get update \
|
||||
&& ACCEPT_EULA=Y apt-get install -y --no-install-recommends \
|
||||
msodbcsql17 \
|
||||
unixodbc-dev \
|
||||
&& apt-get clean autoclean \
|
||||
&& apt-get autoremove -y \
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
&& rm -f /var/cache/apt/archives/*.deb
|
||||
|
||||
COPY fetch_gpg_keys.sh /tmp
|
||||
# Install the desired versions of Python.
|
||||
RUN set -ex \
|
||||
&& export GNUPGHOME="$(mktemp -d)" \
|
||||
&& echo "disable-ipv6" >> "${GNUPGHOME}/dirmngr.conf" \
|
||||
&& /tmp/fetch_gpg_keys.sh \
|
||||
&& for PYTHON_VERSION in 2.7.18 3.6.13 3.7.10 3.8.8 3.9.2; do \
|
||||
wget --no-check-certificate -O python-${PYTHON_VERSION}.tar.xz "https://www.python.org/ftp/python/${PYTHON_VERSION%%[a-z]*}/Python-$PYTHON_VERSION.tar.xz" \
|
||||
&& wget --no-check-certificate -O python-${PYTHON_VERSION}.tar.xz.asc "https://www.python.org/ftp/python/${PYTHON_VERSION%%[a-z]*}/Python-$PYTHON_VERSION.tar.xz.asc" \
|
||||
&& gpg --batch --verify python-${PYTHON_VERSION}.tar.xz.asc python-${PYTHON_VERSION}.tar.xz \
|
||||
&& rm -r python-${PYTHON_VERSION}.tar.xz.asc \
|
||||
&& mkdir -p /usr/src/python-${PYTHON_VERSION} \
|
||||
&& tar -xJC /usr/src/python-${PYTHON_VERSION} --strip-components=1 -f python-${PYTHON_VERSION}.tar.xz \
|
||||
&& rm python-${PYTHON_VERSION}.tar.xz \
|
||||
&& cd /usr/src/python-${PYTHON_VERSION} \
|
||||
&& ./configure \
|
||||
--enable-shared \
|
||||
# This works only on Python 2.7 and throws a warning on every other
|
||||
# version, but seems otherwise harmless.
|
||||
--enable-unicode=ucs4 \
|
||||
--with-system-ffi \
|
||||
--without-ensurepip \
|
||||
&& make -j$(nproc) \
|
||||
&& make install \
|
||||
&& ldconfig \
|
||||
; done \
|
||||
&& rm -rf "${GNUPGHOME}" \
|
||||
&& rm -rf /usr/src/python* \
|
||||
&& rm -rf ~/.cache/
|
||||
|
||||
|
||||
# Install pip on Python 3.6 only.
|
||||
# If the environment variable is called "PIP_VERSION", pip explodes with
|
||||
# "ValueError: invalid truth value '<VERSION>'"
|
||||
ENV PYTHON_PIP_VERSION 20.2.4
|
||||
RUN wget --no-check-certificate -O /tmp/get-pip.py 'https://bootstrap.pypa.io/get-pip.py' \
|
||||
&& python3.6 /tmp/get-pip.py "pip==$PYTHON_PIP_VERSION" \
|
||||
# we use "--force-reinstall" for the case where the version of pip we're trying to install is the same as the version bundled with Python
|
||||
# ("Requirement already up-to-date: pip==8.1.2 in /usr/local/lib/python3.6/site-packages")
|
||||
# https://github.com/docker-library/python/pull/143#issuecomment-241032683
|
||||
&& pip3 install --no-cache-dir --upgrade --force-reinstall "pip==$PYTHON_PIP_VERSION" \
|
||||
# then we use "pip list" to ensure we don't have more than one pip version installed
|
||||
# https://github.com/docker-library/python/pull/100
|
||||
&& [ "$(pip list |tac|tac| awk -F '[ ()]+' '$1 == "pip" { print $2; exit }')" = "$PYTHON_PIP_VERSION" ]
|
||||
|
||||
# Ensure Pip for python3
|
||||
RUN python3 /tmp/get-pip.py
|
||||
RUN rm /tmp/get-pip.py
|
||||
|
||||
# Install "virtualenv", since the vast majority of users of this image
|
||||
# will want it.
|
||||
RUN pip install --no-cache-dir virtualenv
|
||||
|
||||
# Setup Cloud SDK
|
||||
ENV CLOUD_SDK_VERSION 339.0.0
|
||||
# Use system python for cloud sdk.
|
||||
ENV CLOUDSDK_PYTHON python3.6
|
||||
RUN wget https://dl.google.com/dl/cloudsdk/channels/rapid/downloads/google-cloud-sdk-$CLOUD_SDK_VERSION-linux-x86_64.tar.gz
|
||||
RUN tar xzf google-cloud-sdk-$CLOUD_SDK_VERSION-linux-x86_64.tar.gz
|
||||
RUN /google-cloud-sdk/install.sh
|
||||
ENV PATH /google-cloud-sdk/bin:$PATH
|
||||
|
||||
# Enable redis-server on boot.
|
||||
RUN sudo systemctl enable redis-server.service
|
||||
|
||||
# Create a user and allow sudo
|
||||
|
||||
# kbuilder uid on the default Kokoro image
|
||||
ARG UID=1000
|
||||
ARG USERNAME=kbuilder
|
||||
|
||||
# Add a new user to the container image.
|
||||
# This is needed for ssh and sudo access.
|
||||
|
||||
# Add a new user with the caller's uid and the username.
|
||||
RUN useradd -d /h -u ${UID} ${USERNAME}
|
||||
|
||||
# Allow nopasswd sudo
|
||||
RUN echo "${USERNAME} ALL=(ALL) NOPASSWD:ALL" >> /etc/sudoers
|
||||
|
||||
CMD ["python3.6"]
|
||||
@@ -1,42 +1,45 @@
|
||||
from typing import List
|
||||
from resource_cleanup_manager import (
|
||||
ResourceCleanupManager,
|
||||
DatasetResourceCleanupManager,
|
||||
EndpointResourceCleanupManager,
|
||||
ModelResourceCleanupManager,
|
||||
ResourceCleanupManager,
|
||||
DatasetResourceCleanupManager,
|
||||
EndpointResourceCleanupManager,
|
||||
ModelResourceCleanupManager,
|
||||
)
|
||||
|
||||
|
||||
def run_cleanup_managers(managers: List[ResourceCleanupManager], is_dry_run: bool):
|
||||
for manager in managers:
|
||||
type_name = manager.type_name
|
||||
for manager in managers:
|
||||
type_name = manager.type_name
|
||||
|
||||
print(f"Fetching {type_name}'s...")
|
||||
resources = manager.list()
|
||||
print(f"Found {len(resources)} {type_name}'s")
|
||||
for resource in resources:
|
||||
if not manager.is_deletable(resource):
|
||||
continue
|
||||
print(f"Fetching {type_name}'s...")
|
||||
resources = manager.list()
|
||||
print(f"Found {len(resources)} {type_name}'s")
|
||||
for resource in resources:
|
||||
if not manager.is_deletable(resource):
|
||||
continue
|
||||
|
||||
if is_dry_run:
|
||||
resource_name = manager.resource_name(resource)
|
||||
print(f"Will delete '{type_name}': {resource_name}")
|
||||
else:
|
||||
manager.delete(resource)
|
||||
if is_dry_run:
|
||||
resource_name = manager.resource_name(resource)
|
||||
print(f"Will delete '{type_name}': {resource_name}")
|
||||
else:
|
||||
try:
|
||||
manager.delete(resource)
|
||||
except Exception as exception:
|
||||
print(exception)
|
||||
|
||||
print("")
|
||||
print("")
|
||||
|
||||
|
||||
is_dry_run = False
|
||||
|
||||
if is_dry_run:
|
||||
print("Starting cleanup in dry run mode...")
|
||||
print("Starting cleanup in dry run mode...")
|
||||
|
||||
# List of all cleanup managers
|
||||
managers = [
|
||||
DatasetResourceCleanupManager(),
|
||||
EndpointResourceCleanupManager(),
|
||||
ModelResourceCleanupManager(),
|
||||
DatasetResourceCleanupManager(),
|
||||
EndpointResourceCleanupManager(),
|
||||
ModelResourceCleanupManager(),
|
||||
]
|
||||
|
||||
run_cleanup_managers(managers=managers, is_dry_run=is_dry_run)
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
#!/usr/bin/env python
|
||||
# Copyright 2021 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""A CLI to process changed notebooks and execute them on Google Cloud Build"""
|
||||
|
||||
import argparse
|
||||
import pathlib
|
||||
import execute_changed_notebooks_helper
|
||||
|
||||
|
||||
def str2bool(v):
|
||||
if isinstance(v, bool):
|
||||
return v
|
||||
if v.lower() in ("yes", "true", "t", "y", "1"):
|
||||
return True
|
||||
elif v.lower() in ("no", "false", "f", "n", "0"):
|
||||
return False
|
||||
else:
|
||||
raise argparse.ArgumentTypeError("Boolean value expected.")
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
parser.add_argument(
|
||||
"--test_paths_file",
|
||||
type=pathlib.Path,
|
||||
help="The path to the file that has newline-limited folders of notebooks that should be tested.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--base_branch",
|
||||
help="The base git branch to diff against to find changed files.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container_uri",
|
||||
type=str,
|
||||
help="The container uri to run each notebook in.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_project_id",
|
||||
type=str,
|
||||
help="The GCP project id. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_region",
|
||||
type=str,
|
||||
help="The GCP region. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--staging_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for staging temporary files.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifacts_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for storing executed notebooks.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--private_pool_id",
|
||||
type=str,
|
||||
help="The private pool id.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--should_parallelize",
|
||||
type=str2bool,
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=True,
|
||||
help="Should run notebooks in parallel.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
notebooks = execute_changed_notebooks_helper.get_changed_notebooks(
|
||||
test_paths_file=args.test_paths_file,
|
||||
base_branch=args.base_branch,
|
||||
)
|
||||
|
||||
execute_changed_notebooks_helper.process_and_execute_notebooks(
|
||||
notebooks=notebooks,
|
||||
container_uri=args.container_uri,
|
||||
staging_bucket=args.staging_bucket,
|
||||
artifacts_bucket=args.artifacts_bucket,
|
||||
variable_project_id=args.variable_project_id,
|
||||
variable_region=args.variable_region,
|
||||
private_pool_id=args.private_pool_id if not "default" else None,
|
||||
should_parallelize=args.should_parallelize,
|
||||
)
|
||||
@@ -13,7 +13,6 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import concurrent
|
||||
import dataclasses
|
||||
import datetime
|
||||
@@ -32,17 +31,6 @@ from utils import util, NotebookProcessors
|
||||
from google.cloud.devtools.cloudbuild_v1.types import BuildOperationMetadata
|
||||
|
||||
|
||||
def str2bool(v):
|
||||
if isinstance(v, bool):
|
||||
return v
|
||||
if v.lower() in ("yes", "true", "t", "y", "1"):
|
||||
return True
|
||||
elif v.lower() in ("no", "false", "f", "n", "0"):
|
||||
return False
|
||||
else:
|
||||
raise argparse.ArgumentTypeError("Boolean value expected.")
|
||||
|
||||
|
||||
def format_timedelta(delta: datetime.timedelta) -> str:
|
||||
"""Formats a timedelta duration to [N days] %H:%M:%S format"""
|
||||
seconds = int(delta.total_seconds())
|
||||
@@ -115,12 +103,13 @@ def _create_tag(filepath: str) -> str:
|
||||
return tag
|
||||
|
||||
|
||||
def execute_notebook(
|
||||
def process_and_execute_notebook(
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
private_pool_id: Optional[str],
|
||||
notebook: str,
|
||||
should_get_tail_logs: bool = False,
|
||||
) -> NotebookExecutionResult:
|
||||
@@ -162,6 +151,8 @@ def execute_notebook(
|
||||
notebook_output_uri=notebook_output_uri,
|
||||
container_uri=container_uri,
|
||||
tag=tag,
|
||||
region=variable_region,
|
||||
private_pool_id=private_pool_id,
|
||||
)
|
||||
|
||||
operation_metadata = BuildOperationMetadata(mapping=operation.metadata)
|
||||
@@ -202,41 +193,13 @@ def execute_notebook(
|
||||
return result
|
||||
|
||||
|
||||
def run_changed_notebooks(
|
||||
def get_changed_notebooks(
|
||||
test_paths_file: str,
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
should_parallelize: bool,
|
||||
base_branch: Optional[str] = None,
|
||||
):
|
||||
) -> List[str]:
|
||||
"""
|
||||
Run the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only runs notebooks that have differences from the Git base_branch.
|
||||
|
||||
The executed notebooks are saved in the artifacts_bucket.
|
||||
|
||||
Variables are also injected into the notebooks such as the variable_project_id and variable_region.
|
||||
|
||||
Args:
|
||||
test_paths_file (str):
|
||||
Required. The new-line delimited file to folders and files that need checking.
|
||||
Folders are checked recursively.
|
||||
base_branch (str):
|
||||
Optional. If provided, only the files that have changed from the base_branch will be checked.
|
||||
If not provided, all files will be checked.
|
||||
staging_bucket (str):
|
||||
Required. The GCS staging bucket to write source code to.
|
||||
artifacts_bucket (str):
|
||||
Required. The GCS staging bucket to write executed notebooks to.
|
||||
variable_project_id (str):
|
||||
Required. The value for PROJECT_ID to inject into notebooks.
|
||||
variable_region (str):
|
||||
Required. The value for REGION to inject into notebooks.
|
||||
should_parallelize (bool):
|
||||
Required. Should run notebooks in parallel using a thread pool as opposed to in sequence.
|
||||
Get the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only returns notebooks that have differences from the Git base_branch.
|
||||
"""
|
||||
|
||||
test_paths = []
|
||||
@@ -266,6 +229,45 @@ def run_changed_notebooks(
|
||||
notebooks = [notebook for notebook in notebooks if len(notebook) > 0]
|
||||
notebooks = [notebook for notebook in notebooks if pathlib.Path(notebook).exists()]
|
||||
|
||||
return notebooks
|
||||
|
||||
|
||||
def process_and_execute_notebooks(
|
||||
notebooks: List[str],
|
||||
container_uri: str,
|
||||
staging_bucket: str,
|
||||
artifacts_bucket: str,
|
||||
variable_project_id: str,
|
||||
variable_region: str,
|
||||
private_pool_id: Optional[str],
|
||||
should_parallelize: bool,
|
||||
):
|
||||
"""
|
||||
Run the notebooks that exist under the folders defined in the test_paths_file.
|
||||
It only runs notebooks that have differences from the Git base_branch.
|
||||
|
||||
The executed notebooks are saved in the artifacts_bucket.
|
||||
|
||||
Variables are also injected into the notebooks such as the variable_project_id and variable_region.
|
||||
|
||||
Args:
|
||||
test_paths_file (str):
|
||||
Required. The new-line delimited file to folders and files that need checking.
|
||||
Folders are checked recursively.
|
||||
base_branch (str):
|
||||
Optional. If provided, only the files that have changed from the base_branch will be checked.
|
||||
If not provided, all files will be checked.
|
||||
staging_bucket (str):
|
||||
Required. The GCS staging bucket to write source code to.
|
||||
artifacts_bucket (str):
|
||||
Required. The GCS staging bucket to write executed notebooks to.
|
||||
variable_project_id (str):
|
||||
Required. The value for PROJECT_ID to inject into notebooks.
|
||||
variable_region (str):
|
||||
Required. The value for REGION to inject into notebooks.
|
||||
should_parallelize (bool):
|
||||
Required. Should run notebooks in parallel using a thread pool as opposed to in sequence.
|
||||
"""
|
||||
notebook_execution_results: List[NotebookExecutionResult] = []
|
||||
|
||||
if len(notebooks) > 0:
|
||||
@@ -279,24 +281,26 @@ def run_changed_notebooks(
|
||||
notebook_execution_results = list(
|
||||
executor.map(
|
||||
functools.partial(
|
||||
execute_notebook,
|
||||
process_and_execute_notebook,
|
||||
container_uri,
|
||||
staging_bucket,
|
||||
artifacts_bucket,
|
||||
variable_project_id,
|
||||
variable_region,
|
||||
private_pool_id,
|
||||
),
|
||||
notebooks,
|
||||
)
|
||||
)
|
||||
else:
|
||||
notebook_execution_results = [
|
||||
execute_notebook(
|
||||
process_and_execute_notebook(
|
||||
container_uri=container_uri,
|
||||
staging_bucket=staging_bucket,
|
||||
artifacts_bucket=artifacts_bucket,
|
||||
variable_project_id=variable_project_id,
|
||||
variable_region=variable_region,
|
||||
private_pool_id=private_pool_id,
|
||||
notebook=notebook,
|
||||
)
|
||||
for notebook in notebooks
|
||||
@@ -341,67 +345,3 @@ def run_changed_notebooks(
|
||||
# Raise error if any notebooks failed
|
||||
if not all([result.is_pass for result in results_sorted]):
|
||||
raise RuntimeError("Notebook failures detected. See logs for details")
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
parser.add_argument(
|
||||
"--test_paths_file",
|
||||
type=pathlib.Path,
|
||||
help="The path to the file that has newline-limited folders of notebooks that should be tested.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--base_branch",
|
||||
help="The base git branch to diff against to find changed files.",
|
||||
required=False,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--container_uri",
|
||||
type=str,
|
||||
help="The container uri to run each notebook in.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_project_id",
|
||||
type=str,
|
||||
help="The GCP project id. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--variable_region",
|
||||
type=str,
|
||||
help="The GCP region. This is used to inject a variable value into the notebook before running.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--staging_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for staging temporary files.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--artifacts_bucket",
|
||||
type=str,
|
||||
help="The GCP directory for storing executed notebooks.",
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument(
|
||||
"--should_parallelize",
|
||||
type=str2bool,
|
||||
nargs="?",
|
||||
const=True,
|
||||
default=True,
|
||||
help="Should run notebooks in parallel.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
run_changed_notebooks(
|
||||
test_paths_file=args.test_paths_file,
|
||||
container_uri=args.container_uri,
|
||||
staging_bucket=args.staging_bucket,
|
||||
artifacts_bucket=args.artifacts_bucket,
|
||||
variable_project_id=args.variable_project_id,
|
||||
variable_region=args.variable_region,
|
||||
should_parallelize=args.should_parallelize,
|
||||
base_branch=args.base_branch,
|
||||
)
|
||||
@@ -13,10 +13,12 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import argparse
|
||||
import ExecuteNotebook
|
||||
"""A CLI to download (optional) and run a single notebook locally"""
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run changed notebooks.")
|
||||
import argparse
|
||||
import execute_notebook_helper
|
||||
|
||||
parser = argparse.ArgumentParser(description="Run a single notebook locally.")
|
||||
parser.add_argument(
|
||||
"--notebook_source",
|
||||
type=str,
|
||||
@@ -31,7 +33,7 @@ parser.add_argument(
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
ExecuteNotebook.execute_notebook(
|
||||
execute_notebook_helper.execute_notebook(
|
||||
notebook_source=args.notebook_source,
|
||||
output_file_or_uri=args.output_file_or_uri,
|
||||
should_log_output=True,
|
||||
|
||||
@@ -13,6 +13,8 @@
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Methods to run a notebook locally"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import errno
|
||||
@@ -30,6 +32,7 @@ def execute_notebook(
|
||||
output_file_or_uri: str,
|
||||
should_log_output: bool,
|
||||
):
|
||||
"""Execute a single notebook using Papermill"""
|
||||
file_name = os.path.basename(os.path.normpath(notebook_source))
|
||||
|
||||
# Download notebook if it's a GCS URI
|
||||
@@ -1,3 +1,21 @@
|
||||
#!/usr/bin/env python
|
||||
# Copyright 2021 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Methods to run a notebook on Google Cloud Build"""
|
||||
|
||||
from re import sub
|
||||
from google.protobuf import duration_pb2
|
||||
from yaml.loader import FullLoader
|
||||
|
||||
@@ -9,10 +27,12 @@ from typing import Optional
|
||||
import yaml
|
||||
|
||||
from google.cloud.aiplatform import utils
|
||||
from google.api_core import operation
|
||||
from google.api_core import operation, client_options
|
||||
|
||||
|
||||
CLOUD_BUILD_FILEPATH = ".cloud-build/notebook-execution-test-cloudbuild-single.yaml"
|
||||
TIMEOUT_IN_SECONDS = 86400
|
||||
SERVICE_BASE_PATH = "cloudbuild.googleapis.com"
|
||||
|
||||
|
||||
def execute_notebook_remote(
|
||||
@@ -20,20 +40,12 @@ def execute_notebook_remote(
|
||||
notebook_uri: str,
|
||||
notebook_output_uri: str,
|
||||
container_uri: str,
|
||||
region: str,
|
||||
private_pool_id: Optional[str],
|
||||
tag: Optional[str],
|
||||
) -> operation.Operation:
|
||||
"""Create and execute a simple Google Cloud Build configuration,
|
||||
print the in-progress status and print the completed status."""
|
||||
|
||||
# Authorize the client with Google defaults
|
||||
credentials, project_id = google.auth.default()
|
||||
client = cloudbuild_v1.services.cloud_build.CloudBuildClient()
|
||||
|
||||
build = cloudbuild_v1.Build()
|
||||
|
||||
# The following build steps will output "hello world"
|
||||
# For more information on build configuration, see
|
||||
# https://cloud.google.com/build/docs/configuring-builds/create-basic-configuration
|
||||
"""Create and execute a single notebook on Google Cloud Build"""
|
||||
# Load build steps from YAML
|
||||
cloudbuild_config = yaml.load(open(CLOUD_BUILD_FILEPATH), Loader=FullLoader)
|
||||
|
||||
substitutions = {
|
||||
@@ -42,6 +54,23 @@ def execute_notebook_remote(
|
||||
"_NOTEBOOK_OUTPUT_GCS_URI": notebook_output_uri,
|
||||
}
|
||||
|
||||
build = cloudbuild_v1.Build()
|
||||
|
||||
options: Optional[client_options.ClientOptions] = None
|
||||
if private_pool_id:
|
||||
substitutions["_PRIVATE_POOL_NAME"] = private_pool_id
|
||||
build.options = cloudbuild_config["options"]
|
||||
|
||||
# Switch to the regional endpoint of the pool
|
||||
options = client_options.ClientOptions(
|
||||
api_endpoint=f"{region}-{SERVICE_BASE_PATH}"
|
||||
)
|
||||
|
||||
# Authorize the client with Google defaults
|
||||
credentials, project_id = google.auth.default()
|
||||
|
||||
client = cloudbuild_v1.services.cloud_build.CloudBuildClient(client_options=options)
|
||||
|
||||
(
|
||||
source_archived_file_gcs_bucket,
|
||||
source_archived_file_gcs_object,
|
||||
|
||||
@@ -26,3 +26,6 @@ steps:
|
||||
env:
|
||||
- 'IS_TESTING=1'
|
||||
timeout: 86400s
|
||||
options:
|
||||
pool:
|
||||
name: ${_PRIVATE_POOL_NAME}
|
||||
@@ -5,10 +5,6 @@ steps:
|
||||
args:
|
||||
- -c
|
||||
- 'gcloud config list'
|
||||
# # Clone the Git repo
|
||||
# - name: ${_PYTHON_IMAGE}
|
||||
# entrypoint: git
|
||||
# args: ['clone', "${_GIT_REPO}", "--branch", "${_GIT_BRANCH_NAME}", "."]
|
||||
# Check the Python version
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
@@ -28,11 +24,15 @@ steps:
|
||||
- -c
|
||||
- 'python3 -m pip install -U pip && python3 -m pip install -U --user -r .cloud-build/requirements.txt'
|
||||
# Install Python dependencies and run testing script
|
||||
# TODO: Only pass in private_pool_id if it is set
|
||||
- name: ${_PYTHON_IMAGE}
|
||||
entrypoint: /bin/sh
|
||||
args:
|
||||
- -c
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/ExecuteChangedNotebooks.py --test_paths_file "${_TEST_PATHS_FILE}" --base_branch "${_FORCED_BASE_BRANCH}" --container_uri ${_PYTHON_IMAGE} --staging_bucket ${_GCS_STAGING_BUCKET} --artifacts_bucket ${_GCS_STAGING_BUCKET}/executed_notebooks/PR_${_PR_NUMBER}/BUILD_${BUILD_ID} --variable_project_id ${PROJECT_ID} --variable_region ${_GCP_REGION}'
|
||||
- 'python3 -m pip install -U pip && python3 -m pip freeze && python3 .cloud-build/execute_changed_notebooks_cli.py --test_paths_file "${_TEST_PATHS_FILE}" --base_branch "${_FORCED_BASE_BRANCH}" --container_uri ${_PYTHON_IMAGE} --staging_bucket ${_GCS_STAGING_BUCKET} --artifacts_bucket ${_GCS_STAGING_BUCKET}/executed_notebooks/PR_${_PR_NUMBER}/BUILD_${BUILD_ID} --variable_project_id ${PROJECT_ID} --variable_region ${_GCP_REGION} `if [ ! -z "${_PRIVATE_POOL_NAME}" ]; then echo "--private_pool_id ${_PRIVATE_POOL_NAME}"; fi`'
|
||||
env:
|
||||
- 'IS_TESTING=1'
|
||||
timeout: 86400s
|
||||
options:
|
||||
pool:
|
||||
name: ${_PRIVATE_POOL_NAME}
|
||||
@@ -1,10 +1,10 @@
|
||||
ipython==7.30.1
|
||||
jupyter==1.0.0
|
||||
nbconvert==6.3.0
|
||||
papermill==2.3.3
|
||||
numpy==1.21.4
|
||||
pandas==1.3.5
|
||||
matplotlib==3.5.1
|
||||
ipython
|
||||
numpy
|
||||
jupyter
|
||||
nbconvert
|
||||
papermill
|
||||
pandas
|
||||
matplotlib
|
||||
tabulate
|
||||
google-cloud-aiplatform
|
||||
google-cloud-storage
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
notebooks/official/vizier/gapic-vizier-multi-objective-optimization.ipynb
|
||||
notebooks/official/pipelines/lightweight_functions_component_io_kfp.ipynb
|
||||
notebooks/official/matching_engine/intro-swivel.ipynb
|
||||
notebooks/official/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb
|
||||
notebooks/official/pipelines/metrics_viz_run_compare_kfp.ipynb
|
||||
@@ -11,7 +11,7 @@ import uuid
|
||||
|
||||
|
||||
def download_file(bucket_name: str, blob_name: str, destination_file: str) -> str:
|
||||
"""Copies a remote GCS file to a local path."""
|
||||
"""Copies a remote GCS file to a local path"""
|
||||
remote_file_path = "".join(["gs://", "/".join([bucket_name, blob_name])])
|
||||
|
||||
subprocess.check_output(
|
||||
@@ -25,7 +25,7 @@ def upload_file(
|
||||
local_file_path: str,
|
||||
remote_file_path: str,
|
||||
) -> str:
|
||||
"""Copies a local file to a GCS path."""
|
||||
"""Copies a local file to a GCS path"""
|
||||
subprocess.check_output(
|
||||
["gsutil", "cp", local_file_path, remote_file_path], encoding="UTF-8"
|
||||
)
|
||||
|
||||
@@ -7,9 +7,9 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v2
|
||||
uses: actions/setup-python@v3
|
||||
- name: Fetch pull request branch
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Fetch base main branch
|
||||
|
||||
@@ -2,8 +2,8 @@ git+https://github.com/tensorflow/docs
|
||||
ipython
|
||||
jupyter
|
||||
nbconvert
|
||||
black==21.10b0
|
||||
pyupgrade==2.29.1
|
||||
black==22.1.0
|
||||
pyupgrade==2.31.1
|
||||
isort==5.10.1
|
||||
flake8==4.0.1
|
||||
nbqa==1.2.2
|
||||
nbqa==1.3.1
|
||||
|
||||
@@ -84,19 +84,19 @@ if [ ${#notebooks[@]} -gt 0 ]; then
|
||||
FLAKE8_RTN=$?
|
||||
else
|
||||
echo "Running black..."
|
||||
python3 -m nbqa black "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa black "$notebook"
|
||||
BLACK_RTN=$?
|
||||
echo "Running pyupgrade..."
|
||||
python3 -m nbqa pyupgrade "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa pyupgrade "$notebook"
|
||||
PYUPGRADE_RTN=$?
|
||||
echo "Running isort..."
|
||||
python3 -m nbqa isort "$notebook" --nbqa-mutate
|
||||
python3 -m nbqa isort "$notebook"
|
||||
ISORT_RTN=$?
|
||||
echo "Running nbfmt..."
|
||||
python3 -m tensorflow_docs.tools.nbfmt --remove_outputs "$notebook"
|
||||
NBFMT_RTN=$?
|
||||
echo "Running flake8..."
|
||||
python3 -m nbqa flake8 "$notebook" --show-source --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291 --nbqa-mutate
|
||||
python3 -m nbqa flake8 "$notebook" --show-source --extend-ignore=W391,E501,F821,E402,F404,W503,E203,E722,W293,W291
|
||||
FLAKE8_RTN=$?
|
||||
fi
|
||||
|
||||
|
||||
@@ -31,7 +31,7 @@ pip3 install --user -U nbqa black flake8 isort pyupgrade git+https://github.com/
|
||||
You'll likely need to add the directory where these were installed to your PATH:
|
||||
|
||||
```shell
|
||||
export PATH=“$HOME/.local/bin:$PATH"
|
||||
export PATH="$HOME/.local/bin:$PATH"
|
||||
```
|
||||
|
||||
Then, set an environment variable for your notebook (or directory):
|
||||
|
||||
@@ -6,7 +6,19 @@ Welcome to the Google Cloud [Vertex AI](https://cloud.google.com/vertex-ai/docs/
|
||||
|
||||
## Overview
|
||||
|
||||
The repository contains [Notebooks](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks) and [Community Content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/community-content) that demonstrate how to develop and manage ML workflows using Google Cloud Vertex AI.
|
||||
The repository contains [notebooks](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks) and [community content](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/community-content) that demonstrate how to develop and manage ML workflows using Google Cloud Vertex AI.
|
||||
|
||||
## Repository structure
|
||||
|
||||
```bash
|
||||
├── community-content - Sample code and tutorials contributed by the community
|
||||
├── notebooks
|
||||
│ ├── community - Notebooks contributed by the community
|
||||
│ ├── official - Notebooks demonstrating use of each Vertex AI service
|
||||
│ │ ├── automl
|
||||
│ │ ├── custom
|
||||
│ │ ├── ...
|
||||
```
|
||||
|
||||
## Contributing
|
||||
|
||||
@@ -19,3 +31,7 @@ Please use the [issues page](https://github.com/GoogleCloudPlatform/vertex-ai-sa
|
||||
## Disclaimer
|
||||
|
||||
This is not an officially supported Google product. The code in this repository is for demonstrative purposes only.
|
||||
|
||||
## Feedback
|
||||
|
||||
Please feel free to fill out our [survey](https://bit.ly/vertex-ai-samples-survey) to give us feedback on the repo and its content.
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
* @vertex-ai-samples-contributors @GoogleCloudPlatform/cloudml-samples-owners
|
||||
/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk @yinghsienwu
|
||||
/pytorch_text_classification_using_vertex_sdk_and_gcloud @RajeshThallam
|
||||
/pytorch_text_classification_using_vertex_sdk_and_gcloud @RajeshThallam @ultrons
|
||||
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
|
||||
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
|
||||
|
||||
@@ -0,0 +1,824 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "pc5-mbsX9PZC"
|
||||
},
|
||||
"source": [
|
||||
"# AlphaFold On Vertex AI Workbench\n",
|
||||
"\n",
|
||||
"[Vertex AI Workbench](https://cloud.google.com/vertex-ai/docs/workbench) offers an end-to-end notebook-based production environment that can be preconfigured with the runtime dependencies necessary to run AlphaFold on Vertex AI. With [User-Managed Notebooks](https://cloud.google.com/vertex-ai/docs/workbench/user-managed/introduction), you can configure a GPU accelerator to run AlphaFold using Tensorflow, without having to install and manage drivers or JupyterLab instances. This notebook allows you to easily predict the structure of a protein using a slightly simplified version of [AlphaFold v2.1.0](https://doi.org/10.1038/s41586-021-03819-2). \n",
|
||||
"\n",
|
||||
"##  [Launch this Notebook in Vertex AI Workbench](https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/raw/main/community-content/alphafold_on_workbench/AlphaFold.ipynb)\n",
|
||||
"\n",
|
||||
"**Differences to AlphaFold v2.1.0**\n",
|
||||
"\n",
|
||||
"In comparison to AlphaFold v2.1.0, this notebook notebook uses **no templates (homologous structures)** and a selected portion of the [BFD database](https://bfd.mmseqs.com/). We have validated these changes on several thousand recent PDB structures. While accuracy will be near-identical to the full AlphaFold system on many targets, a small fraction have a large drop in accuracy due to the smaller MSA and lack of templates. For best reliability, we recommend instead using the [full open source AlphaFold](https://github.com/deepmind/alphafold/), or the [AlphaFold Protein Structure Database](https://alphafold.ebi.ac.uk/).\n",
|
||||
"\n",
|
||||
"**This notebook has an small drop in average accuracy for multimers compared to local AlphaFold installation, for full multimer accuracy it is highly recommended to run [AlphaFold locally](https://github.com/deepmind/alphafold#running-alphafold).** Moreover, the AlphaFold-Multimer requires searching for MSA for every unique sequence in the complex, hence it is substantially slower. If your notebook times-out due to slow multimer MSA search, we recommend running AlphaFold locally.\n",
|
||||
"\n",
|
||||
"Please note that this notebook is provided as an early-access prototype and is not a finished product. It is provided for theoretical modelling only and caution should be exercised in its use. \n",
|
||||
"\n",
|
||||
"**Citing this work**\n",
|
||||
"\n",
|
||||
"Any publication that discloses findings arising from using this notebook should [cite](https://github.com/deepmind/alphafold/#citing-this-work) the [AlphaFold paper](https://doi.org/10.1038/s41586-021-03819-2).\n",
|
||||
"\n",
|
||||
"**Licenses**\n",
|
||||
"\n",
|
||||
"This Colab uses the [AlphaFold model parameters](https://github.com/deepmind/alphafold/#model-parameters-license) which are subject to the Creative Commons Attribution 4.0 International ([CC BY 4.0](https://creativecommons.org/licenses/by/4.0/legalcode)) license. The Colab itself is provided under the [Apache 2.0 license](https://www.apache.org/licenses/LICENSE-2.0). See the full license statement below.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"**More information**\n",
|
||||
"\n",
|
||||
"You can find more information about how AlphaFold works in the following papers:\n",
|
||||
"\n",
|
||||
"* [AlphaFold methods paper](https://www.nature.com/articles/s41586-021-03819-2)\n",
|
||||
"* [AlphaFold predictions of the human proteome paper](https://www.nature.com/articles/s41586-021-03828-1)\n",
|
||||
"* [AlphaFold-Multimer paper](https://www.biorxiv.org/content/10.1101/2021.10.04.463034v1)\n",
|
||||
"\n",
|
||||
"FAQ on how to interpret AlphaFold predictions are [here](https://alphafold.ebi.ac.uk/faq)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "b7a02613eb1a"
|
||||
},
|
||||
"source": [
|
||||
"## Download AlphaFold Data"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "woIxeCPygt7K"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"import subprocess\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"import alphafold.common\n",
|
||||
"import tqdm.notebook\n",
|
||||
"from IPython.utils import io\n",
|
||||
"\n",
|
||||
"TQDM_BAR_FORMAT = (\n",
|
||||
" \"{l_bar}{bar}| {n_fmt}/{total_fmt} [elapsed: {elapsed} remaining: {remaining}]\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"SOURCE_URL = (\n",
|
||||
" \"https://storage.googleapis.com/alphafold/alphafold_params_colab_2022-01-19.tar\"\n",
|
||||
")\n",
|
||||
"PARAMS_DIR = \"alphafold/data/params\"\n",
|
||||
"PARAMS_PATH = os.path.join(PARAMS_DIR, os.path.basename(SOURCE_URL))\n",
|
||||
"ALPHAFOLD_COMMON_DIR = os.path.dirname(alphafold.common.__file__)\n",
|
||||
"\n",
|
||||
"try:\n",
|
||||
" with tqdm.notebook.tqdm(total=100, bar_format=TQDM_BAR_FORMAT) as pbar:\n",
|
||||
" with io.capture_output() as captured:\n",
|
||||
"\n",
|
||||
" # Download and store stereo_chemical_props.txt\n",
|
||||
" !mkdir -p ~/content/alphafold/alphafold/common\n",
|
||||
" !mkdir -p /opt/conda/lib/python3.7/site-packages/alphafold/common/\n",
|
||||
" !wget -q -P ~/content/alphafold/alphafold/common https://git.scicore.unibas.ch/schwede/openstructure/-/raw/7102c63615b64735c4941278d92b554ec94415f8/modules/mol/alg/src/stereo_chemical_props.txt\n",
|
||||
" pbar.update(18)\n",
|
||||
" !cp -f ~/content/alphafold/alphafold/common/stereo_chemical_props.txt \"{ALPHAFOLD_COMMON_DIR}\"\n",
|
||||
"\n",
|
||||
" # Download alphafold_params_colab_2021-10-27.tar\n",
|
||||
" !mkdir --parents \"{PARAMS_DIR}\"\n",
|
||||
" !wget -O \"{PARAMS_PATH}\" \"{SOURCE_URL}\"\n",
|
||||
" pbar.update(27)\n",
|
||||
"\n",
|
||||
" # Un-tar alphafold_params_colab_2021-10-27.tar\n",
|
||||
" !tar --extract --verbose --file=\"{PARAMS_PATH}\" --directory=\"{PARAMS_DIR}\" --preserve-permissions\n",
|
||||
" # !rm \"{PARAMS_PATH}\"\n",
|
||||
" pbar.update(55)\n",
|
||||
"\n",
|
||||
"except subprocess.CalledProcessError:\n",
|
||||
" print(captured)\n",
|
||||
" raise"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d8926b7d5529"
|
||||
},
|
||||
"source": [
|
||||
"## Configure GPU Acceleration"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "VzJ5iMjTtoZw"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Confirm accelerator configuration\n",
|
||||
"import jax\n",
|
||||
"\n",
|
||||
"if jax.local_devices()[0].platform == \"tpu\":\n",
|
||||
" raise RuntimeError(\n",
|
||||
" \"TPU runtime not supported. Please configure GPU acceleration on the VM.\"\n",
|
||||
" )\n",
|
||||
"elif jax.local_devices()[0].platform == \"cpu\":\n",
|
||||
" print(\n",
|
||||
" \"CPU-only runtime is not recommended, because prediction execution will be slow. For better performance, consider GPU acceleration on the VM.\"\n",
|
||||
" )\n",
|
||||
"else:\n",
|
||||
" print(f\"Running with {jax.local_devices()[0].device_kind} GPU\")\n",
|
||||
"\n",
|
||||
"# Make sure all necessary environment variables are set.\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"os.environ[\"TF_FORCE_UNIFIED_MEMORY\"] = \"1\"\n",
|
||||
"os.environ[\"XLA_PYTHON_CLIENT_MEM_FRACTION\"] = \"2.0\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "W4JpOs6oA-QS"
|
||||
},
|
||||
"source": [
|
||||
"## Making a prediction\n",
|
||||
"\n",
|
||||
"Please paste the sequence of your protein in the text box below, then run the remaining cells via _Run_ > _Run Selected Cell and All Below_. You can also run the cells individually by pressing the _Play_ button on the left.\n",
|
||||
"\n",
|
||||
"Note that the search against databases and the actual prediction can take some time, from minutes to hours, depending on the length of the protein and what type of GPU you allocate (see FAQ below).\n",
|
||||
"\n",
|
||||
"To start, enter the amino acid sequence(s) to fold ⬇️\n",
|
||||
"\n",
|
||||
"If you enter only a single sequence, the monomer model will be used. If you enter multiple sequences, the multimer model will be used."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b310d44229d0"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Input sequences (type: str)\n",
|
||||
"sequence_1 = \"MAAHKGAEHHHKAAEHHEQAAKHHHAAAEHHEKGEHEQAAHHADTAYAHHKHAEEHAAQAAKHDAEHHAPKPH\"\n",
|
||||
"sequence_2 = \"\"\n",
|
||||
"sequence_3 = \"\"\n",
|
||||
"sequence_4 = \"\"\n",
|
||||
"sequence_5 = \"\"\n",
|
||||
"sequence_6 = \"\"\n",
|
||||
"sequence_7 = \"\"\n",
|
||||
"sequence_8 = \"\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "rowN0bVYLe9n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from alphafold.notebooks import notebook_utils\n",
|
||||
"\n",
|
||||
"input_sequences = (\n",
|
||||
" sequence_1,\n",
|
||||
" sequence_2,\n",
|
||||
" sequence_3,\n",
|
||||
" sequence_4,\n",
|
||||
" sequence_5,\n",
|
||||
" sequence_6,\n",
|
||||
" sequence_7,\n",
|
||||
" sequence_8,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# If folding a complex target and all the input sequences are\n",
|
||||
"# prokaryotic then set `is_prokaryotic` to `True`. Set to `False`\n",
|
||||
"# otherwise or if the origin is unknown.\n",
|
||||
"\n",
|
||||
"is_prokaryote = False # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"MIN_SINGLE_SEQUENCE_LENGTH = 16\n",
|
||||
"MAX_SINGLE_SEQUENCE_LENGTH = 2500\n",
|
||||
"MAX_MULTIMER_LENGTH = 2500\n",
|
||||
"\n",
|
||||
"# Validate the input.\n",
|
||||
"sequences, model_type_to_use = notebook_utils.validate_input(\n",
|
||||
" input_sequences=input_sequences,\n",
|
||||
" min_length=MIN_SINGLE_SEQUENCE_LENGTH,\n",
|
||||
" max_length=MAX_SINGLE_SEQUENCE_LENGTH,\n",
|
||||
" max_multimer_length=MAX_MULTIMER_LENGTH,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "db551d4877ea"
|
||||
},
|
||||
"source": [
|
||||
"## Search against genetic databases\n",
|
||||
"\n",
|
||||
"Once this cell has been executed, you will see statistics about the multiple sequence alignment (MSA) that will be used by AlphaFold. In particular, you’ll see how well each residue is covered by similar sequences in the MSA."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "2tTeTTsLKPjB"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import collections\n",
|
||||
"import copy\n",
|
||||
"import random\n",
|
||||
"from concurrent import futures\n",
|
||||
"from urllib import request\n",
|
||||
"\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import numpy as np\n",
|
||||
"import py3Dmol\n",
|
||||
"from alphafold.common import protein\n",
|
||||
"from alphafold.data import (feature_processing, msa_pairing, pipeline,\n",
|
||||
" pipeline_multimer)\n",
|
||||
"from alphafold.data.tools import jackhmmer\n",
|
||||
"from alphafold.model import config, data, model\n",
|
||||
"from alphafold.relax import relax, utils\n",
|
||||
"from IPython import display\n",
|
||||
"from ipywidgets import GridspecLayout, Output\n",
|
||||
"\n",
|
||||
"# Color bands for visualizing plddt\n",
|
||||
"PLDDT_BANDS = [\n",
|
||||
" (0, 50, \"#FF7D45\"),\n",
|
||||
" (50, 70, \"#FFDB13\"),\n",
|
||||
" (70, 90, \"#65CBF3\"),\n",
|
||||
" (90, 100, \"#0053D6\"),\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"# --- Find the closest source ---\n",
|
||||
"test_url_pattern = (\n",
|
||||
" \"https://storage.googleapis.com/alphafold-colab{:s}/latest/uniref90_2021_03.fasta.1\"\n",
|
||||
")\n",
|
||||
"ex = futures.ThreadPoolExecutor(3)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def fetch(source):\n",
|
||||
" request.urlretrieve(test_url_pattern.format(source))\n",
|
||||
" return source\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"fs = [ex.submit(fetch, source) for source in [\"\", \"-europe\", \"-asia\"]]\n",
|
||||
"source = None\n",
|
||||
"for f in futures.as_completed(fs):\n",
|
||||
" source = f.result()\n",
|
||||
" ex.shutdown()\n",
|
||||
" break\n",
|
||||
"\n",
|
||||
"JACKHMMER_BINARY_PATH = \"/usr/bin/jackhmmer\"\n",
|
||||
"DB_ROOT_PATH = f\"https://storage.googleapis.com/alphafold-colab{source}/latest/\"\n",
|
||||
"# The z_value is the number of sequences in a database.\n",
|
||||
"MSA_DATABASES = [\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"uniref90\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}uniref90_2021_03.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 59,\n",
|
||||
" \"z_value\": 135_301_051,\n",
|
||||
" },\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"smallbfd\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}bfd-first_non_consensus_sequences.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 17,\n",
|
||||
" \"z_value\": 65_984_053,\n",
|
||||
" },\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"mgnify\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}mgy_clusters_2019_05.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 71,\n",
|
||||
" \"z_value\": 304_820_129,\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"# Search UniProt and construct the all_seq features only for heteromers, not homomers.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MULTIMER and len(set(sequences)) > 1:\n",
|
||||
" MSA_DATABASES.extend(\n",
|
||||
" [\n",
|
||||
" # Swiss-Prot and TrEMBL are concatenated together as UniProt.\n",
|
||||
" {\n",
|
||||
" \"db_name\": \"uniprot\",\n",
|
||||
" \"db_path\": f\"{DB_ROOT_PATH}uniprot_2021_03.fasta\",\n",
|
||||
" \"num_streamed_chunks\": 98,\n",
|
||||
" \"z_value\": 219_174_961 + 565_254,\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"TOTAL_JACKHMMER_CHUNKS = sum(cfg[\"num_streamed_chunks\"] for cfg in MSA_DATABASES)\n",
|
||||
"\n",
|
||||
"MAX_HITS = {\n",
|
||||
" \"uniref90\": 10_000,\n",
|
||||
" \"smallbfd\": 5_000,\n",
|
||||
" \"mgnify\": 501,\n",
|
||||
" \"uniprot\": 50_000,\n",
|
||||
"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_msa(fasta_path):\n",
|
||||
" \"\"\"Searches for MSA for the given sequence using chunked Jackhmmer search.\"\"\"\n",
|
||||
"\n",
|
||||
" # Run the search against chunks of genetic databases.\n",
|
||||
" raw_msa_results = collections.defaultdict(list)\n",
|
||||
" with tqdm.notebook.tqdm(\n",
|
||||
" total=TOTAL_JACKHMMER_CHUNKS, bar_format=TQDM_BAR_FORMAT\n",
|
||||
" ) as pbar:\n",
|
||||
"\n",
|
||||
" def jackhmmer_chunk_callback(i):\n",
|
||||
" pbar.update(n=1)\n",
|
||||
"\n",
|
||||
" for db_config in MSA_DATABASES:\n",
|
||||
" db_name = db_config[\"db_name\"]\n",
|
||||
" pbar.set_description(f\"Searching {db_name}\")\n",
|
||||
" jackhmmer_runner = jackhmmer.Jackhmmer(\n",
|
||||
" binary_path=JACKHMMER_BINARY_PATH,\n",
|
||||
" database_path=db_config[\"db_path\"],\n",
|
||||
" get_tblout=True,\n",
|
||||
" num_streamed_chunks=db_config[\"num_streamed_chunks\"],\n",
|
||||
" streaming_callback=jackhmmer_chunk_callback,\n",
|
||||
" z_value=db_config[\"z_value\"],\n",
|
||||
" )\n",
|
||||
" # Group the results by database name.\n",
|
||||
" raw_msa_results[db_name].extend(jackhmmer_runner.query(fasta_path))\n",
|
||||
"\n",
|
||||
" return raw_msa_results\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"features_for_chain = {}\n",
|
||||
"raw_msa_results_for_sequence = {}\n",
|
||||
"for sequence_index, sequence in enumerate(sequences, start=1):\n",
|
||||
" print(f\"\\nGetting MSA for sequence {sequence_index}\")\n",
|
||||
"\n",
|
||||
" fasta_path = f\"target_{sequence_index}.fasta\"\n",
|
||||
" with open(fasta_path, \"wt\") as f:\n",
|
||||
" f.write(f\">query\\n{sequence}\")\n",
|
||||
"\n",
|
||||
" # Don't do redundant work for multiple copies of the same chain in the multimer.\n",
|
||||
" if sequence not in raw_msa_results_for_sequence:\n",
|
||||
" raw_msa_results = get_msa(fasta_path=fasta_path)\n",
|
||||
" raw_msa_results_for_sequence[sequence] = raw_msa_results\n",
|
||||
" else:\n",
|
||||
" raw_msa_results = copy.deepcopy(raw_msa_results_for_sequence[sequence])\n",
|
||||
"\n",
|
||||
" # Extract the MSAs from the Stockholm files.\n",
|
||||
" # NB: deduplication happens later in pipeline.make_msa_features.\n",
|
||||
" single_chain_msas = []\n",
|
||||
" uniprot_msa = None\n",
|
||||
" for db_name, db_results in raw_msa_results.items():\n",
|
||||
" merged_msa = notebook_utils.merge_chunked_msa(\n",
|
||||
" results=db_results, max_hits=MAX_HITS.get(db_name)\n",
|
||||
" )\n",
|
||||
" if merged_msa.sequences and db_name != \"uniprot\":\n",
|
||||
" single_chain_msas.append(merged_msa)\n",
|
||||
" msa_size = len(set(merged_msa.sequences))\n",
|
||||
" print(\n",
|
||||
" f\"{msa_size} unique sequences found in {db_name} for sequence {sequence_index}\"\n",
|
||||
" )\n",
|
||||
" elif merged_msa.sequences and db_name == \"uniprot\":\n",
|
||||
" uniprot_msa = merged_msa\n",
|
||||
"\n",
|
||||
" notebook_utils.show_msa_info(\n",
|
||||
" single_chain_msas=single_chain_msas, sequence_index=sequence_index\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Turn the raw data into model features.\n",
|
||||
" feature_dict = {}\n",
|
||||
" feature_dict.update(\n",
|
||||
" pipeline.make_sequence_features(\n",
|
||||
" sequence=sequence, description=\"query\", num_res=len(sequence)\n",
|
||||
" )\n",
|
||||
" )\n",
|
||||
" feature_dict.update(pipeline.make_msa_features(msas=single_chain_msas))\n",
|
||||
" # We don't use templates in AlphaFold notebook, add only empty placeholder features.\n",
|
||||
" feature_dict.update(\n",
|
||||
" notebook_utils.empty_placeholder_template_features(\n",
|
||||
" num_templates=0, num_res=len(sequence)\n",
|
||||
" )\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Construct the all_seq features only for heteromers, not homomers.\n",
|
||||
" if (\n",
|
||||
" model_type_to_use == notebook_utils.ModelType.MULTIMER\n",
|
||||
" and len(set(sequences)) > 1\n",
|
||||
" ):\n",
|
||||
" valid_feats = msa_pairing.MSA_FEATURES + (\n",
|
||||
" \"msa_uniprot_accession_identifiers\",\n",
|
||||
" \"msa_species_identifiers\",\n",
|
||||
" )\n",
|
||||
" all_seq_features = {\n",
|
||||
" f\"{k}_all_seq\": v\n",
|
||||
" for k, v in pipeline.make_msa_features([uniprot_msa]).items()\n",
|
||||
" if k in valid_feats\n",
|
||||
" }\n",
|
||||
" feature_dict.update(all_seq_features)\n",
|
||||
"\n",
|
||||
" features_for_chain[protein.PDB_CHAIN_IDS[sequence_index - 1]] = feature_dict\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Do further feature post-processing depending on the model type.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" np_example = features_for_chain[protein.PDB_CHAIN_IDS[0]]\n",
|
||||
"\n",
|
||||
"elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" all_chain_features = {}\n",
|
||||
" for chain_id, chain_features in features_for_chain.items():\n",
|
||||
" all_chain_features[chain_id] = pipeline_multimer.convert_monomer_features(\n",
|
||||
" chain_features, chain_id\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" all_chain_features = pipeline_multimer.add_assembly_features(all_chain_features)\n",
|
||||
"\n",
|
||||
" np_example = feature_processing.pair_and_merge(\n",
|
||||
" all_chain_features=all_chain_features, is_prokaryote=is_prokaryote\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Pad MSA to avoid zero-sized extra_msa.\n",
|
||||
" np_example = pipeline_multimer.pad_msa(np_example, min_num_seq=512)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "9640643486bd"
|
||||
},
|
||||
"source": [
|
||||
"## Run AlphaFold\n",
|
||||
"\n",
|
||||
"Once this cell has been executed, a zip-archive \"prediction.zip\" with the obtained prediction will be saved on the VM, and available for download to your computer in the sidebar. In case you are having issues with the relaxation stage, you can disable it below. Warning: This means that the prediction might have distracting small stereochemical violations."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": true,
|
||||
"jupyter": {
|
||||
"source_hidden": true
|
||||
},
|
||||
"cellView": "form",
|
||||
"id": "XUo6foMQxwS2"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"run_relax = True\n",
|
||||
"\n",
|
||||
"# --- Run the model ---\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" model_names = config.MODEL_PRESETS[\"monomer\"] + (\"model_2_ptm\",)\n",
|
||||
"elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" model_names = config.MODEL_PRESETS[\"multimer\"]\n",
|
||||
"\n",
|
||||
"output_dir = \"prediction\"\n",
|
||||
"os.makedirs(output_dir, exist_ok=True)\n",
|
||||
"\n",
|
||||
"plddts = {}\n",
|
||||
"ranking_confidences = {}\n",
|
||||
"pae_outputs = {}\n",
|
||||
"unrelaxed_proteins = {}\n",
|
||||
"\n",
|
||||
"with tqdm.notebook.tqdm(total=len(model_names) + 1, bar_format=TQDM_BAR_FORMAT) as pbar:\n",
|
||||
" for model_name in model_names:\n",
|
||||
" pbar.set_description(f\"Running {model_name}\")\n",
|
||||
"\n",
|
||||
" cfg = config.model_config(model_name)\n",
|
||||
" if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" cfg.data.eval.num_ensemble = 1\n",
|
||||
" elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" cfg.model.num_ensemble_eval = 1\n",
|
||||
" params = data.get_model_haiku_params(model_name, \"./alphafold/data\")\n",
|
||||
" model_runner = model.RunModel(cfg, params)\n",
|
||||
" processed_feature_dict = model_runner.process_features(\n",
|
||||
" np_example, random_seed=0\n",
|
||||
" )\n",
|
||||
" prediction = model_runner.predict(\n",
|
||||
" processed_feature_dict, random_seed=random.randrange(sys.maxsize)\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" mean_plddt = prediction[\"plddt\"].mean()\n",
|
||||
"\n",
|
||||
" if model_type_to_use == notebook_utils.ModelType.MONOMER:\n",
|
||||
" if \"predicted_aligned_error\" in prediction:\n",
|
||||
" pae_outputs[model_name] = (\n",
|
||||
" prediction[\"predicted_aligned_error\"],\n",
|
||||
" prediction[\"max_predicted_aligned_error\"],\n",
|
||||
" )\n",
|
||||
" else:\n",
|
||||
" # Monomer models are sorted by mean pLDDT. Do not put monomer pTM models here as they\n",
|
||||
" # should never get selected.\n",
|
||||
" ranking_confidences[model_name] = prediction[\"ranking_confidence\"]\n",
|
||||
" plddts[model_name] = prediction[\"plddt\"]\n",
|
||||
" elif model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" # Multimer models are sorted by pTM+ipTM.\n",
|
||||
" ranking_confidences[model_name] = prediction[\"ranking_confidence\"]\n",
|
||||
" plddts[model_name] = prediction[\"plddt\"]\n",
|
||||
" pae_outputs[model_name] = (\n",
|
||||
" prediction[\"predicted_aligned_error\"],\n",
|
||||
" prediction[\"max_predicted_aligned_error\"],\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Set the b-factors to the per-residue plddt.\n",
|
||||
" final_atom_mask = prediction[\"structure_module\"][\"final_atom_mask\"]\n",
|
||||
" b_factors = prediction[\"plddt\"][:, None] * final_atom_mask\n",
|
||||
" unrelaxed_protein = protein.from_prediction(\n",
|
||||
" processed_feature_dict,\n",
|
||||
" prediction,\n",
|
||||
" b_factors=b_factors,\n",
|
||||
" remove_leading_feature_dimension=(\n",
|
||||
" model_type_to_use == notebook_utils.ModelType.MONOMER\n",
|
||||
" ),\n",
|
||||
" )\n",
|
||||
" unrelaxed_proteins[model_name] = unrelaxed_protein\n",
|
||||
"\n",
|
||||
" # Delete unused outputs to save memory.\n",
|
||||
" del model_runner\n",
|
||||
" del params\n",
|
||||
" del prediction\n",
|
||||
" pbar.update(n=1)\n",
|
||||
"\n",
|
||||
" # --- AMBER relax the best model ---\n",
|
||||
"\n",
|
||||
" # Find the best model according to the mean pLDDT.\n",
|
||||
" best_model_name = max(\n",
|
||||
" ranking_confidences.keys(), key=lambda x: ranking_confidences[x]\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" if run_relax:\n",
|
||||
" pbar.set_description(\"AMBER relaxation\")\n",
|
||||
" amber_relaxer = relax.AmberRelaxation(\n",
|
||||
" max_iterations=0,\n",
|
||||
" tolerance=2.39,\n",
|
||||
" stiffness=10.0,\n",
|
||||
" exclude_residues=[],\n",
|
||||
" max_outer_iterations=3,\n",
|
||||
" )\n",
|
||||
" relaxed_pdb, _, _ = amber_relaxer.process(\n",
|
||||
" prot=unrelaxed_proteins[best_model_name]\n",
|
||||
" )\n",
|
||||
" else:\n",
|
||||
" print(\"Warning: Running without the relaxation stage.\")\n",
|
||||
" relaxed_pdb = protein.to_pdb(unrelaxed_proteins[best_model_name])\n",
|
||||
" pbar.update(n=1) # Finished AMBER relax.\n",
|
||||
"\n",
|
||||
"# Construct multiclass b-factors to indicate confidence bands\n",
|
||||
"# 0=very low, 1=low, 2=confident, 3=very high\n",
|
||||
"banded_b_factors = []\n",
|
||||
"for plddt in plddts[best_model_name]:\n",
|
||||
" for idx, (min_val, max_val, _) in enumerate(PLDDT_BANDS):\n",
|
||||
" if plddt >= min_val and plddt <= max_val:\n",
|
||||
" banded_b_factors.append(idx)\n",
|
||||
" break\n",
|
||||
"banded_b_factors = np.array(banded_b_factors)[:, None] * final_atom_mask\n",
|
||||
"to_visualize_pdb = utils.overwrite_b_factors(relaxed_pdb, banded_b_factors)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Write out the prediction\n",
|
||||
"pred_output_path = os.path.join(output_dir, \"selected_prediction.pdb\")\n",
|
||||
"with open(pred_output_path, \"w\") as f:\n",
|
||||
" f.write(relaxed_pdb)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# --- Visualise the prediction & confidence ---\n",
|
||||
"show_sidechains = True\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def plot_plddt_legend():\n",
|
||||
" \"\"\"Plots the legend for pLDDT.\"\"\"\n",
|
||||
" thresh = [\n",
|
||||
" \"Very low (pLDDT < 50)\",\n",
|
||||
" \"Low (70 > pLDDT > 50)\",\n",
|
||||
" \"Confident (90 > pLDDT > 70)\",\n",
|
||||
" \"Very high (pLDDT > 90)\",\n",
|
||||
" ]\n",
|
||||
"\n",
|
||||
" colors = [x[2] for x in PLDDT_BANDS]\n",
|
||||
"\n",
|
||||
" plt.figure(figsize=(2, 2))\n",
|
||||
" for c in colors:\n",
|
||||
" plt.bar(0, 0, color=c)\n",
|
||||
" plt.legend(thresh, frameon=False, loc=\"center\", fontsize=20)\n",
|
||||
" plt.xticks([])\n",
|
||||
" plt.yticks([])\n",
|
||||
" ax = plt.gca()\n",
|
||||
" ax.spines[\"right\"].set_visible(False)\n",
|
||||
" ax.spines[\"top\"].set_visible(False)\n",
|
||||
" ax.spines[\"left\"].set_visible(False)\n",
|
||||
" ax.spines[\"bottom\"].set_visible(False)\n",
|
||||
" plt.title(\"Model Confidence\", fontsize=20, pad=20)\n",
|
||||
" return plt\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Show the structure coloured by chain if the multimer model has been used.\n",
|
||||
"if model_type_to_use == notebook_utils.ModelType.MULTIMER:\n",
|
||||
" multichain_view = py3Dmol.view(width=800, height=600)\n",
|
||||
" multichain_view.addModelsAsFrames(to_visualize_pdb)\n",
|
||||
" multichain_style = {\"cartoon\": {\"colorscheme\": \"chain\"}}\n",
|
||||
" multichain_view.setStyle({\"model\": -1}, multichain_style)\n",
|
||||
" multichain_view.zoomTo()\n",
|
||||
" multichain_view.show()\n",
|
||||
"\n",
|
||||
"# Color the structure by per-residue pLDDT\n",
|
||||
"color_map = {i: bands[2] for i, bands in enumerate(PLDDT_BANDS)}\n",
|
||||
"view = py3Dmol.view(width=800, height=600)\n",
|
||||
"view.addModelsAsFrames(to_visualize_pdb)\n",
|
||||
"style = {\"cartoon\": {\"colorscheme\": {\"prop\": \"b\", \"map\": color_map}}}\n",
|
||||
"if show_sidechains:\n",
|
||||
" style[\"stick\"] = {}\n",
|
||||
"view.setStyle({\"model\": -1}, style)\n",
|
||||
"view.zoomTo()\n",
|
||||
"\n",
|
||||
"grid = GridspecLayout(1, 2)\n",
|
||||
"out = Output()\n",
|
||||
"with out:\n",
|
||||
" view.show()\n",
|
||||
"grid[0, 0] = out\n",
|
||||
"\n",
|
||||
"out = Output()\n",
|
||||
"with out:\n",
|
||||
" plot_plddt_legend().show()\n",
|
||||
"grid[0, 1] = out\n",
|
||||
"\n",
|
||||
"display.display(grid)\n",
|
||||
"\n",
|
||||
"# Display pLDDT and predicted aligned error (if output by the model).\n",
|
||||
"if pae_outputs:\n",
|
||||
" num_plots = 2\n",
|
||||
"else:\n",
|
||||
" num_plots = 1\n",
|
||||
"\n",
|
||||
"plt.figure(figsize=[8 * num_plots, 6])\n",
|
||||
"plt.subplot(1, num_plots, 1)\n",
|
||||
"plt.plot(plddts[best_model_name])\n",
|
||||
"plt.title(\"Predicted LDDT\")\n",
|
||||
"plt.xlabel(\"Residue\")\n",
|
||||
"plt.ylabel(\"pLDDT\")\n",
|
||||
"\n",
|
||||
"if num_plots == 2:\n",
|
||||
" plt.subplot(1, 2, 2)\n",
|
||||
" pae, max_pae = list(pae_outputs.values())[0]\n",
|
||||
" plt.imshow(pae, vmin=0.0, vmax=max_pae, cmap=\"Greens_r\")\n",
|
||||
" plt.colorbar(fraction=0.046, pad=0.04)\n",
|
||||
"\n",
|
||||
" # Display lines at chain boundaries.\n",
|
||||
" best_unrelaxed_prot = unrelaxed_proteins[best_model_name]\n",
|
||||
" total_num_res = best_unrelaxed_prot.residue_index.shape[-1]\n",
|
||||
" chain_ids = best_unrelaxed_prot.chain_index\n",
|
||||
" for chain_boundary in np.nonzero(chain_ids[:-1] - chain_ids[1:]):\n",
|
||||
" if chain_boundary.size:\n",
|
||||
" plt.plot([0, total_num_res], [chain_boundary, chain_boundary], color=\"red\")\n",
|
||||
" plt.plot([chain_boundary, chain_boundary], [0, total_num_res], color=\"red\")\n",
|
||||
"\n",
|
||||
" plt.title(\"Predicted Aligned Error\")\n",
|
||||
" plt.xlabel(\"Scored residue\")\n",
|
||||
" plt.ylabel(\"Aligned residue\")\n",
|
||||
"\n",
|
||||
"# Save the predicted aligned error (if it exists).\n",
|
||||
"pae_output_path = os.path.join(output_dir, \"predicted_aligned_error.json\")\n",
|
||||
"if pae_outputs:\n",
|
||||
" # Save predicted aligned error in the same format as the AF EMBL DB.\n",
|
||||
" pae_data = notebook_utils.get_pae_json(pae=pae, max_pae=max_pae.item())\n",
|
||||
" with open(pae_output_path, \"w\") as f:\n",
|
||||
" f.write(pae_data)\n",
|
||||
"\n",
|
||||
"!zip -q -r {output_dir}.zip {output_dir}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "lUQAn5LYC5n4"
|
||||
},
|
||||
"source": [
|
||||
"### Interpreting the prediction\n",
|
||||
"\n",
|
||||
"In general predicted LDDT (pLDDT) is best used for intra-domain confidence, whereas Predicted Aligned Error (PAE) is best used for determining between domain or between chain confidence.\n",
|
||||
"\n",
|
||||
"Please see the [AlphaFold methods paper](https://www.nature.com/articles/s41586-021-03819-2), the [AlphaFold predictions of the human proteome paper](https://www.nature.com/articles/s41586-021-03828-1), and the [AlphaFold-Multimer paper](https://www.biorxiv.org/content/10.1101/2021.10.04.463034v1) as well as [our FAQ](https://alphafold.ebi.ac.uk/faq) on how to interpret AlphaFold predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "jeb2z8DIA4om"
|
||||
},
|
||||
"source": [
|
||||
"## FAQ & Troubleshooting\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"* How do I get a predicted protein structure for my protein?\n",
|
||||
" * Connect the notebook to the Jupyter kernel \"Python 3 (ipykernel)\".\n",
|
||||
" * Paste the amino acid sequence of your protein (without any headers) into the variable sequence_1 in \"Making a Prediction\".\n",
|
||||
" * Run all cells in the notebook, either by running them individually or via \"Kernel\"/\"Restart Kernel and Run All Cells...\"\n",
|
||||
" * The predicted protein structure will be downloaded once all cells have been executed. Note: This can take minutes to hours - see below.\n",
|
||||
"* How long will this take?\n",
|
||||
" * The search against genetic databases can take minutes to hours.\n",
|
||||
" * Running AlphaFold and generating the prediction can take minutes to hours, depending on the length of your protein and on which GPU-type your VM has access to.\n",
|
||||
"* My notebook no longer seems to be doing anything, what should I do?\n",
|
||||
" * Some steps may take minutes to hours to complete.\n",
|
||||
" * If nothing happens or if you receive an error message, try restarting your notebook runtime via \"Kernel\"/\"Restart Kernel and Run All Cells...\".\n",
|
||||
" * If this doesn’t help, try resetting restarting your VM inside the GCloud Console (\"Compute Engine\"/\"VM Instances\").\n",
|
||||
"* How does this compare to the open-source version of AlphaFold?\n",
|
||||
" * This notebook version of AlphaFold searches a selected portion of the BFD dataset and currently doesn’t use templates, so its accuracy is reduced in comparison to the full version of AlphaFold that is described in the [AlphaFold paper](https://doi.org/10.1038/s41586-021-03819-2) and [Github repo](https://github.com/deepmind/alphafold/) (the full version is available via the inference script).\n",
|
||||
"* I received a warning “Notebook requires high RAM”, what do I do?\n",
|
||||
" * In the \"Compute Engine\"/\"VM Instances\" Console menu, you can reconfigure the host VM settings. See [Changing the machine type of a VM instance](https://cloud.google.com/compute/docs/instances/changing-machine-type-of-stopped-instance) for instructions.\n",
|
||||
"* Does this tool install anything on my computer?\n",
|
||||
" * No, everything happens in the VM instance within your Google Cloud project.\n",
|
||||
"* How should I share feedback and bug reports?\n",
|
||||
" * Please share any feedback and bug reports as an [issue](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues) on Github.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Related work\n",
|
||||
"\n",
|
||||
"Take a look at these Colab notebooks provided by the community (please note that these notebooks may vary from our validated AlphaFold system and we cannot guarantee their accuracy):\n",
|
||||
"\n",
|
||||
"* The [ColabFold AlphaFold2 notebook](https://colab.research.google.com/github/sokrypton/ColabFold/blob/main/AlphaFold2.ipynb) by Sergey Ovchinnikov, Milot Mirdita and Martin Steinegger, which uses an API hosted at the Södinglab based on the MMseqs2 server ([Mirdita et al. 2019, Bioinformatics](https://academic.oup.com/bioinformatics/article/35/16/2856/5280135)) for the multiple sequence alignment creation.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "YfPhvYgKC81B"
|
||||
},
|
||||
"source": [
|
||||
"# License and Disclaimer\n",
|
||||
"\n",
|
||||
"This is not an officially-supported Google product.\n",
|
||||
"\n",
|
||||
"This notebook and other information provided is for theoretical modelling only, caution should be exercised in its use. It is provided ‘as-is’ without any warranty of any kind, whether expressed or implied. Information is not intended to be a substitute for professional medical advice, diagnosis, or treatment, and does not constitute medical or other professional advice.\n",
|
||||
"\n",
|
||||
"Copyright 2021 DeepMind Technologies Limited.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## AlphaFold Code License\n",
|
||||
"\n",
|
||||
"Licensed under the Apache License, Version 2.0 (the \"License\"); you may not use this file except in compliance with the License. You may obtain a copy of the License at https://www.apache.org/licenses/LICENSE-2.0.\n",
|
||||
"\n",
|
||||
"Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on an \"AS IS\" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the specific language governing permissions and limitations under the License.\n",
|
||||
"\n",
|
||||
"## Model Parameters License\n",
|
||||
"\n",
|
||||
"The AlphaFold parameters are made available under the terms of the Creative Commons Attribution 4.0 International (CC BY 4.0) license. You can find details at: https://creativecommons.org/licenses/by/4.0/legalcode\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Third-party software\n",
|
||||
"\n",
|
||||
"Use of the third-party software, libraries or code referred to in the [Acknowledgements section](https://github.com/deepmind/alphafold/#acknowledgements) in the AlphaFold README may be governed by separate terms and conditions or license provisions. Your use of the third-party software, libraries or code is subject to any such terms and you should check that you can comply with any applicable restrictions or terms and conditions before use.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Mirrored Databases\n",
|
||||
"\n",
|
||||
"The following databases have been mirrored by DeepMind, and are available with reference to the following:\n",
|
||||
"* UniProt: v2021\\_03 (unmodified), by The UniProt Consortium, available under a [Creative Commons Attribution-NoDerivatives 4.0 International License](http://creativecommons.org/licenses/by-nd/4.0/).\n",
|
||||
"* UniRef90: v2021\\_03 (unmodified), by The UniProt Consortium, available under a [Creative Commons Attribution-NoDerivatives 4.0 International License](http://creativecommons.org/licenses/by-nd/4.0/).\n",
|
||||
"* MGnify: v2019\\_05 (unmodified), by Mitchell AL et al., available free of all copyright restrictions and made fully and freely available for both non-commercial and commercial use under [CC0 1.0 Universal (CC0 1.0) Public Domain Dedication](https://creativecommons.org/publicdomain/zero/1.0/).\n",
|
||||
"* BFD: (modified), by Steinegger M. and Söding J., modified by DeepMind, available under a [Creative Commons Attribution-ShareAlike 4.0 International License](https://creativecommons.org/licenses/by/4.0/). See the Methods section of the [AlphaFold proteome paper](https://www.nature.com/articles/s41586-021-03828-1) for details."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"accelerator": "GPU",
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "AlphaFold.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
# Copyright 2022 Google LLC
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
ARG CUDA_MAJOR=11
|
||||
ARG CUDA_MINOR=0
|
||||
|
||||
FROM gcr.io/deeplearning-platform-release/base-cu110
|
||||
|
||||
ARG CUDA_MAJOR
|
||||
ARG CUDA_MINOR
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
|
||||
RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
|
||||
build-essential \
|
||||
cmake \
|
||||
cuda-command-line-tools-${CUDA_MAJOR}-${CUDA_MINOR} \
|
||||
git \
|
||||
hmmer \
|
||||
kalign \
|
||||
tzdata \
|
||||
wget \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Compile HHsuite from source.
|
||||
RUN git clone --branch v3.3.0 https://github.com/soedinglab/hh-suite.git /tmp/hh-suite \
|
||||
&& mkdir /tmp/hh-suite/build \
|
||||
&& pushd /tmp/hh-suite/build \
|
||||
&& cmake -DCMAKE_INSTALL_PREFIX=/opt/hhsuite .. \
|
||||
&& make -j 4 && make install \
|
||||
&& ln -s /opt/hhsuite/bin/* /usr/bin \
|
||||
&& popd \
|
||||
&& rm -rf /tmp/hh-suite
|
||||
|
||||
ENV PATH="/opt/conda/bin:$PATH"
|
||||
RUN conda update -qy conda \
|
||||
&& conda install -y -c conda-forge \
|
||||
openmm=7.5.1 \
|
||||
cudatoolkit==${CUDA_VERSION} \
|
||||
pdbfixer \
|
||||
pip \
|
||||
python=3.7
|
||||
|
||||
COPY . /app/alphafold
|
||||
|
||||
# Install pip packages.
|
||||
RUN pip3 install --upgrade pip \
|
||||
&& pip3 install -r /app/alphafold/requirements.txt \
|
||||
&& pip3 install py3Dmol tqdm \
|
||||
&& pip3 install --upgrade jax==0.2.14 jaxlib==0.1.69+cuda${CUDA_MAJOR}${CUDA_MINOR} -f \
|
||||
https://storage.googleapis.com/jax-releases/jax_releases.html
|
||||
|
||||
# Install alphafold.
|
||||
WORKDIR /app/alphafold
|
||||
RUN python setup.py install
|
||||
|
||||
# Apply OpenMM patch.
|
||||
WORKDIR /opt/conda/lib/python3.7/site-packages
|
||||
RUN patch -p0 < /app/alphafold/docker/openmm.patch
|
||||
|
||||
# Creating a tmp location for jackhmmr; not mounting through to host though.
|
||||
RUN sudo mkdir -m 777 --parents /tmp/ramdisk
|
||||
|
||||
# We need to run `ldconfig` first to ensure GPUs are visible, due to some quirk
|
||||
# with Debian. See https://github.com/NVIDIA/nvidia-docker/issues/1399 for
|
||||
# details.
|
||||
# ENTRYPOINT does not support easily running multiple commands, so instead we
|
||||
# write a shell script to wrap them up.
|
||||
WORKDIR /home/jupyter
|
||||
RUN echo '#!/bin/bash\nldconfig\n\'
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
#!/usr/bin/env bash
|
||||
set -e
|
||||
|
||||
# Prod (Publicly viewable)
|
||||
PROJECT=cloud-devrel-public-resources
|
||||
REPOSITORY=alphafold
|
||||
LOCAL_IMAGE=alphafold-on-gcp
|
||||
REMOTE_IMAGE=${LOCAL_IMAGE?}
|
||||
TAG=latest
|
||||
REGISTRY="us-west1-docker.pkg.dev/${PROJECT?}/${REPOSITORY?}/${REMOTE_IMAGE?}:${TAG?}"
|
||||
|
||||
git clone https://github.com/deepmind/alphafold.git
|
||||
|
||||
cp Dockerfile alphafold/docker/Dockerfile
|
||||
cp AlphaFold.ipynb alphafold/notebooks/AlphaFold.ipynb
|
||||
|
||||
cd alphafold && sudo docker build --tag ${LOCAL_IMAGE?}:${TAG?} -f docker/Dockerfile .
|
||||
|
||||
sudo docker tag ${LOCAL_IMAGE?}:${TAG?} ${REGISTRY?}
|
||||
sudo docker push ${REGISTRY?}
|
||||
|
After Width: | Height: | Size: 3.1 KiB |
@@ -1,6 +1,6 @@
|
||||
# PyTorch on Google Cloud: Text Classification
|
||||
|
||||
In the PyTorch on Google Cloud series of blog posts, we aim to share how to build, train and deploy PyTorch models at scale and how to create reproducible machine learning pipelines on Google Cloud with [Vertex AI](https://cloud.google.com/vertex-ai).
|
||||
In the PyTorch on Google Cloud series of blog posts, we aim to share how to build, train, deploy and orchestrate PyTorch models at scale and how to create reproducible machine learning pipelines on Google Cloud with [Vertex AI](https://cloud.google.com/vertex-ai).
|
||||
|
||||
This tutorial on text classification shows how to train a PyTorch based text classification model by fine tuning a pre-trained Huggingface Transformers model and deploy the model on [Vertex AI](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) using Vertex SDK and [`gcloud ai`](https://cloud.google.com/sdk/gcloud/reference/beta/ai).
|
||||
|
||||
@@ -9,6 +9,7 @@ This tutorial on text classification shows how to train a PyTorch based text cla
|
||||
| <h4>Notebook</h4> | <h4>Description</h4> |
|
||||
| :-------- | :------- |
|
||||
| [pytorch-text-classification-vertex-ai-train-tune-deploy.ipynb](./pytorch-text-classification-vertex-ai-train-tune-deploy.ipynb) | Notebook to show training, hyper-parameter tuning and deploying a PyTorch model on Vertex AI |
|
||||
| [pytorch-text-classification-vertex-ai-pipelines.ipynb](./pytorch-text-classification-vertex-ai-pipelines.ipynb) | Notebook to show orchestration of PyTorch ML workflows on Vertex AI Pipelines using Kubeflow Pipelines SDK |
|
||||
|
||||
## Folders
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
|
||||
# Use pytorch GPU base image
|
||||
FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7
|
||||
# FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7
|
||||
FROM us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-10:latest
|
||||
|
||||
# set working directory
|
||||
WORKDIR /app
|
||||
|
||||
@@ -22,15 +22,18 @@ PROJECT_ID=$(gcloud config list --format 'value(core.project)')
|
||||
|
||||
# BUCKET_NAME: Change to your bucket name.
|
||||
BUCKET_NAME="[your-bucket-name]" # <-- CHANGE TO YOUR BUCKET NAME
|
||||
BUCKET_NAME=cloud-ai-platform-2f444b6a-a742-444b-b91a-c7519f51bd77
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# JOB_NAME: the name of your job running on AI Platform.
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-cstm-cntr-"
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-cstm-cntr"
|
||||
JOB_NAME=${JOB_PREFIX}-$(date +%Y%m%d%H%M%S)-custom-job
|
||||
|
||||
# This can be a GCS location to a zipped and uploaded package
|
||||
PACKAGE_PATH=./trainer
|
||||
|
||||
# REGION: select a region from https://cloud.google.com/vertex-ai/docs/general/locations#available_regions
|
||||
# or use the default '`us-central1`'. The region is where the job will be run.
|
||||
REGION="us-central1"
|
||||
@@ -41,11 +44,8 @@ JOB_DIR=gs://${BUCKET_NAME}/${JOB_PREFIX}/models/${JOB_NAME}
|
||||
# IMAGE_REPO_NAME: set a local repo name to distinquish our image
|
||||
IMAGE_REPO_NAME=pytorch_gpu_train_finetuned-bert-classifier
|
||||
|
||||
# IMAGE_TAG: an easily identifiable tag for your docker image
|
||||
IMAGE_TAG=latest
|
||||
|
||||
# IMAGE_URI: the complete URI location for Cloud Container Registry
|
||||
CUSTOM_TRAIN_IMAGE_URI=gcr.io/${PROJECT_ID}/${IMAGE_REPO_NAME}:${IMAGE_TAG}
|
||||
CUSTOM_TRAIN_IMAGE_URI=gcr.io/${PROJECT_ID}/${IMAGE_REPO_NAME}
|
||||
|
||||
# Build the docker image
|
||||
docker build --no-cache -f Dockerfile -t $CUSTOM_TRAIN_IMAGE_URI ../python_package
|
||||
@@ -53,11 +53,19 @@ docker build --no-cache -f Dockerfile -t $CUSTOM_TRAIN_IMAGE_URI ../python_packa
|
||||
# Deploy the docker image to Cloud Container Registry
|
||||
docker push ${CUSTOM_TRAIN_IMAGE_URI}
|
||||
|
||||
# worker pool spec
|
||||
worker_pool_spec="\
|
||||
replica-count=1,\
|
||||
machine-type=n1-standard-8,\
|
||||
accelerator-type=NVIDIA_TESLA_V100,\
|
||||
accelerator-count=1,\
|
||||
container-image-uri=${CUSTOM_TRAIN_IMAGE_URI}"
|
||||
|
||||
# Submit Custom Job to Vertex AI
|
||||
gcloud beta ai custom-jobs create \
|
||||
--display-name=${JOB_NAME} \
|
||||
--region ${REGION} \
|
||||
--worker-pool-spec=replica-count=1,machine-type='n1-standard-8',accelerator-type='NVIDIA_TESLA_V100',accelerator-count=1,container-image-uri=${CUSTOM_TRAIN_IMAGE_URI} \
|
||||
--worker-pool-spec="${worker_pool_spec}" \
|
||||
--args="--model-name","finetuned-bert-classifier","--job-dir",$JOB_DIR
|
||||
|
||||
echo "After the job is completed successfully, model files will be saved at $JOB_DIR/"
|
||||
|
||||
|
After Width: | Height: | Size: 45 KiB |
|
After Width: | Height: | Size: 37 KiB |
|
After Width: | Height: | Size: 76 KiB |
|
After Width: | Height: | Size: 74 KiB |
|
After Width: | Height: | Size: 248 KiB |
|
After Width: | Height: | Size: 38 KiB |
|
After Width: | Height: | Size: 123 KiB |
@@ -2,10 +2,13 @@
|
||||
FROM pytorch/torchserve:latest-cpu
|
||||
|
||||
# install dependencies
|
||||
RUN python3 -m pip install --upgrade pip
|
||||
RUN pip3 install transformers
|
||||
|
||||
USER model-server
|
||||
|
||||
# copy model artifacts, custom handler and other dependencies
|
||||
COPY ./custom_text_handler.py /home/model-server/
|
||||
COPY ./custom_handler.py /home/model-server/
|
||||
COPY ./index_to_name.json /home/model-server/
|
||||
COPY ./model/finetuned-bert-classifier/ /home/model-server/
|
||||
|
||||
@@ -21,7 +24,7 @@ EXPOSE 7080
|
||||
EXPOSE 7081
|
||||
|
||||
# create model archive file packaging model artifacts and dependencies
|
||||
RUN torch-model-archiver -f --model-name=finetuned-bert-classifier --version=1.0 --serialized-file=/home/model-server/pytorch_model.bin --handler=/home/model-server/custom_text_handler.py --extra-files "/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json" --export-path=/home/model-server/model-store
|
||||
RUN torch-model-archiver -f --model-name=finetuned-bert-classifier --version=1.0 --serialized-file=/home/model-server/pytorch_model.bin --handler=/home/model-server/custom_handler.py --extra-files "/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json" --export-path=/home/model-server/model-store
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "finetuned-bert-classifier=finetuned-bert-classifier.mar", "--model-store", "/home/model-server/model-store"]
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
|
||||
FROM pytorch/torchserve:latest-cpu
|
||||
|
||||
USER root
|
||||
# run and update some basic packages software packages, including security libs
|
||||
RUN apt-get update && apt-get install -y software-properties-common && add-apt-repository -y ppa:ubuntu-toolchain-r/test && apt-get update && apt-get install -y gcc-9 g++-9 apt-transport-https ca-certificates gnupg curl
|
||||
|
||||
# Install gcloud tools for gsutil as well as debugging
|
||||
RUN echo "deb [signed-by=/usr/share/keyrings/cloud.google.gpg] http://packages.cloud.google.com/apt cloud-sdk main" | tee -a /etc/apt/sources.list.d/google-cloud-sdk.list && curl https://packages.cloud.google.com/apt/doc/apt-key.gpg | apt-key --keyring /usr/share/keyrings/cloud.google.gpg add - && apt-get update -y && apt-get install google-cloud-sdk -y
|
||||
|
||||
USER model-server
|
||||
|
||||
# install dependencies
|
||||
RUN python3 -m pip install --upgrade pip
|
||||
RUN pip3 install transformers
|
||||
|
||||
ARG MODEL_NAME=finetuned-bert-classifier
|
||||
ENV MODEL_NAME="${MODEL_NAME}"
|
||||
|
||||
# health and prediction listener ports
|
||||
ARG AIP_HTTP_PORT=7080
|
||||
ENV AIP_HTTP_PORT="${AIP_HTTP_PORT}"
|
||||
|
||||
ARG MODEL_MGMT_PORT=7081
|
||||
|
||||
# expose health and prediction listener ports from the image
|
||||
EXPOSE "${AIP_HTTP_PORT}"
|
||||
EXPOSE "${MODEL_MGMT_PORT}"
|
||||
EXPOSE 8080 8081 8082 7070 7071
|
||||
|
||||
# create torchserve configuration file
|
||||
USER root
|
||||
RUN echo "service_envelope=json\n" "inference_address=http://0.0.0.0:${AIP_HTTP_PORT}\n" "management_address=http://0.0.0.0:${MODEL_MGMT_PORT}" >> /home/model-server/config.properties
|
||||
USER model-server
|
||||
|
||||
# run Torchserve HTTP serve to respond to prediction requests
|
||||
CMD ["echo", "AIP_STORAGE_URI=${AIP_STORAGE_URI}", ";", "gsutil", "cp", "-r", "${AIP_STORAGE_URI}/${MODEL_NAME}.mar", "/home/model-server/model-store/", ";", "ls", "-ltr", "/home/model-server/model-store/", ";", "torchserve", "--start", "--ts-config=/home/model-server/config.properties", "--models", "${MODEL_NAME}=${MODEL_NAME}.mar", "--model-store", "/home/model-server/model-store"]
|
||||
@@ -52,7 +52,8 @@ class TransformersClassifierHandler(BaseHandler):
|
||||
with open(mapping_file_path) as f:
|
||||
self.mapping = json.load(f)
|
||||
else:
|
||||
logger.warning('Missing the index_to_name.json file. Inference output will not include class name.')
|
||||
logger.warning('Missing the index_to_name.json file. Inference output will default.')
|
||||
self.mapping = {"0": "Negative", "1": "Positive"}
|
||||
|
||||
self.initialized = True
|
||||
|
||||
@@ -88,4 +89,3 @@ class TransformersClassifierHandler(BaseHandler):
|
||||
|
||||
def postprocess(self, inference_output):
|
||||
return inference_output
|
||||
|
||||
@@ -19,13 +19,19 @@ echo "Submitting Custom Job to Vertex AI to train PyTorch model"
|
||||
|
||||
# BUCKET_NAME: Change to your bucket name
|
||||
BUCKET_NAME="[your-bucket-name]" # <-- CHANGE TO YOUR BUCKET NAME
|
||||
BUCKET_NAME="cloud-ai-platform-2f444b6a-a742-444b-b91a-c7519f51bd77"
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The PyTorch image provided by Vertex AI Training.
|
||||
IMAGE_URI="us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-7:latest"
|
||||
|
||||
# JOB_NAME: the name of your job running on Vertex AI.
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-pkg-ar-"
|
||||
JOB_PREFIX="finetuned-bert-classifier-pytorch-pkg-ar"
|
||||
JOB_NAME=${JOB_PREFIX}-$(date +%Y%m%d%H%M%S)-custom-job
|
||||
|
||||
# REGION: select a region from https://cloud.google.com/vertex-ai/docs/general/locations#available_regions
|
||||
@@ -35,19 +41,21 @@ REGION="us-central1"
|
||||
# JOB_DIR: Where to store prepared package and upload output model.
|
||||
JOB_DIR=gs://${BUCKET_NAME}/${JOB_PREFIX}/model/${JOB_NAME}
|
||||
|
||||
# validate bucket name
|
||||
if [ "${BUCKET_NAME}" = "[your-bucket-name]" ]
|
||||
then
|
||||
echo "[ERROR] INVALID VALUE: Please update the variable BUCKET_NAME with valid Cloud Storage bucket name. Exiting the script..."
|
||||
exit 1
|
||||
fi
|
||||
# worker pool spec
|
||||
worker_pool_spec="\
|
||||
replica-count=1,\
|
||||
machine-type=n1-standard-8,\
|
||||
accelerator-type=NVIDIA_TESLA_V100,\
|
||||
accelerator-count=1,\
|
||||
executor-image-uri=${IMAGE_URI},\
|
||||
python-module=trainer.task,\
|
||||
local-package-path=../python_package/"
|
||||
|
||||
# Submit Custom Job to Vertex AI
|
||||
gcloud beta ai custom-jobs create \
|
||||
--display-name=${JOB_NAME} \
|
||||
--region ${REGION} \
|
||||
--python-package-uris=${PACKAGE_PATH} \
|
||||
--worker-pool-spec=replica-count=1,machine-type='n1-standard-8',accelerator-type='NVIDIA_TESLA_V100',accelerator-count=1,executor-image-uri=${IMAGE_URI},python-module='trainer.task',local-package-path="../python_package/" \
|
||||
--worker-pool-spec="${worker_pool_spec}" \
|
||||
--args="--model-name","finetuned-bert-classifier","--job-dir",$JOB_DIR
|
||||
|
||||
echo "After the job is completed successfully, model files will be saved at $JOB_DIR/"
|
||||
|
||||
@@ -122,6 +122,9 @@ def run(args):
|
||||
# Train / Test the model
|
||||
trainer = train(args, text_classifier, train_dataset, test_dataset)
|
||||
|
||||
metrics = trainer.evaluate(eval_dataset=test_dataset)
|
||||
trainer.save_metrics("all", metrics)
|
||||
|
||||
# Export the trained model
|
||||
trainer.save_model(os.path.join("/tmp", args.model_name))
|
||||
|
||||
|
||||
@@ -63,20 +63,20 @@
|
||||
"- [Training](#Training)\n",
|
||||
" - [Run Training Locally in the Notebook](#Training-locally-in-the-notebook)\n",
|
||||
" - [Run Training Job on Vertex AI](#Training-on-Vertex-AI)\n",
|
||||
" - [Training with pre-built container](#Run-Custom-Job-on-Vertex-Training-with-a-pre-built-container)\n",
|
||||
" - [Training with custom container](#Run-Custom-Job-on-Vertex-Training-with-custom-container)\n",
|
||||
" - [Training with pre-built container](#Run-Custom-Job-on-Vertex-AI-Training-with-a-pre-built-container)\n",
|
||||
" - [Training with custom container](#Run-Custom-Job-on-Vertex-AI-Training-with-custom-container)\n",
|
||||
"- [Tuning](#Hyperparameter-Tuning) \n",
|
||||
" - [Run Hyperparameter Tuning job on Vertex AI](#Run-Hyperparameter-Tuning-Job-on-Vertex-AI)\n",
|
||||
"- [Deploying](#Deploying)\n",
|
||||
" - [Deploying model on Vertex Predictions with custom container](#Deploying-model-on-Vertex-Predictions-with-custom-container)\n",
|
||||
" - [Deploying model on Vertex AI Predictions with custom container](#Deploying-model-on-Vertex AI-Predictions-with-custom-container)\n",
|
||||
"\n",
|
||||
"### Costs \n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud Platform (GCP):\n",
|
||||
"\n",
|
||||
"* [Notebooks](https://cloud.google.com/notebooks)\n",
|
||||
"* [Vertex Training](https://cloud.google.com/vertex-ai/docs/training/custom-training)\n",
|
||||
"* [Vertex Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions)\n",
|
||||
"* [Vertex AI Workbench](https://cloud.google.com/vertex-ai-workbench)\n",
|
||||
"* [Vertex AI Training](https://cloud.google.com/vertex-ai/docs/training/custom-training)\n",
|
||||
"* [Vertex AI Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions)\n",
|
||||
"* [Cloud Storage](https://cloud.google.com/storage)\n",
|
||||
"* [Container Registry](https://cloud.google.com/container-registry)\n",
|
||||
"* [Cloud Build](https://cloud.google.com/build) *[Optional]*\n",
|
||||
@@ -202,9 +202,9 @@
|
||||
"id": "e0c1dcadc2c8"
|
||||
},
|
||||
"source": [
|
||||
"We will be using [Vertex SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) to interact with Vertex AI services. The high-level `aiplatform` library is designed to simplify common data science workflows by using wrapper classes and opinionated defaults. \n",
|
||||
"We will be using [Vertex AI SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#python) to interact with Vertex AI services. The high-level `aiplatform` library is designed to simplify common data science workflows by using wrapper classes and opinionated defaults. \n",
|
||||
"\n",
|
||||
"#### Install Vertex SDK for Python"
|
||||
"#### Install Vertex AI SDK for Python"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1199,7 +1199,7 @@
|
||||
"source": [
|
||||
"### Run predictions locally with sample examples\n",
|
||||
"\n",
|
||||
"Using the trained model, we can predict the sentiment label for an input text after applying the preprocessing function that was used during the training. We will run the predictions locally in the notebook and later show how you can deploy the model to an endpoint using [TorchServe](https://pytorch.org/serve/) on Vertex Predictions."
|
||||
"Using the trained model, we can predict the sentiment label for an input text after applying the preprocessing function that was used during the training. We will run the predictions locally in the notebook and later show how you can deploy the model to an endpoint using [TorchServe](https://pytorch.org/serve/) on Vertex AI Predictions."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1382,7 +1382,7 @@
|
||||
"id": "f7466d414a0e"
|
||||
},
|
||||
"source": [
|
||||
"### Run Custom Job on Vertex Training with a pre-built container"
|
||||
"### Run Custom Job on Vertex AI Training with a pre-built container"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1395,7 +1395,7 @@
|
||||
"\n",
|
||||
"In this notebook, we are using Hugging Face Datasets and fine tuning a transformer model from Hugging Face Transformers Library for sentiment analysis task using PyTorch. We will use [pre-built container for PyTorch](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers#pytorch) and package the training application code by adding standard Python dependencies - `transformers`, `datasets` and `tqdm` - in the `setup.py` file. \n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1569,7 +1569,7 @@
|
||||
"source": [
|
||||
"#### **Run custom training job on Vertex AI**\n",
|
||||
"\n",
|
||||
"We use [Vertex SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#client_libraries) to create and submit training job to the Vertex training service."
|
||||
"We use [Vertex AI SDK for Python](https://cloud.google.com/vertex-ai/docs/start/client-libraries#client_libraries) to create and submit training job to the Vertex AI training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1578,7 +1578,7 @@
|
||||
"id": "5d2957ef04fd"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1598,7 +1598,7 @@
|
||||
"id": "6b0fed34b728"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Custom Job to Vertex Training service**"
|
||||
"##### **Configure and submit Custom Job to Vertex AI Training service**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1609,7 +1609,7 @@
|
||||
"source": [
|
||||
"Configure a [Custom Job](https://cloud.google.com/vertex-ai/docs/training/create-custom-job) with the [pre-built container](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers) image for PyTorch and training code packaged as Python source distribution. \n",
|
||||
"\n",
|
||||
"**NOTE:** When using Vertex SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job on Vertex Training service."
|
||||
"**NOTE:** When using Vertex AI SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job on Vertex AI Training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1686,7 +1686,7 @@
|
||||
"\n",
|
||||
"You can monitor the custom job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/training-pipelines/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1798,7 +1798,7 @@
|
||||
"id": "c170d386492b"
|
||||
},
|
||||
"source": [
|
||||
"### Run Custom Job on Vertex Training with custom container"
|
||||
"### Run Custom Job on Vertex AI Training with custom container"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1807,7 +1807,7 @@
|
||||
"id": "035227b6e581"
|
||||
},
|
||||
"source": [
|
||||
"To create a [training job with custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container?hl=hr), you define a `Dockerfile` to install or add the dependencies required for the training job. Then, you build and test your Docker image locally to verify, push the image to Container Registry and submit a Custom Job to Vertex Training service.\n",
|
||||
"To create a [training job with custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container?hl=hr), you define a `Dockerfile` to install or add the dependencies required for the training job. Then, you build and test your Docker image locally to verify, push the image to Container Registry and submit a Custom Job to Vertex AI Training service.\n",
|
||||
"\n",
|
||||
""
|
||||
]
|
||||
@@ -1834,7 +1834,7 @@
|
||||
"%%writefile ./custom_container/Dockerfile\n",
|
||||
"\n",
|
||||
"# Use pytorch GPU base image\n",
|
||||
"FROM gcr.io/cloud-aiplatform/training/pytorch-gpu.1-7\n",
|
||||
"FROM us-docker.pkg.dev/vertex-ai/training/pytorch-gpu.1-10:latest\n",
|
||||
"\n",
|
||||
"# set working directory\n",
|
||||
"WORKDIR /app\n",
|
||||
@@ -1968,7 +1968,7 @@
|
||||
"id": "a23e5e34bea9"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1988,11 +1988,11 @@
|
||||
"id": "abf1fa4085cb"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Custom Job to Vertex Training service**\n",
|
||||
"##### **Configure and submit Custom Job to Vertex AI Training service**\n",
|
||||
"\n",
|
||||
"Configure a [Custom Job](https://cloud.google.com/vertex-ai/docs/training/create-custom-job) with the [custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container) image with training code and other dependencies\n",
|
||||
"\n",
|
||||
"**NOTE:** When using Vertex SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job to train on Vertex Training."
|
||||
"**NOTE:** When using Vertex AI SDK for Python for submitting a training job, it creates a [Training Pipeline](https://cloud.google.com/vertex-ai/docs/training/create-training-pipeline) which launches the Custom Job to train on Vertex AI Training."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2044,7 +2044,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# submit the custom job to Vertex training service\n",
|
||||
"# submit the custom job to Vertex AI training service\n",
|
||||
"model = job.run(\n",
|
||||
" replica_count=1,\n",
|
||||
" machine_type=\"n1-standard-8\",\n",
|
||||
@@ -2065,7 +2065,7 @@
|
||||
"\n",
|
||||
"You can monitor the custom job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/training-pipelines/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2148,11 +2148,11 @@
|
||||
"id": "ba6122f929e3"
|
||||
},
|
||||
"source": [
|
||||
"The training application code for fine-tuning a transformer model for sentiment analysis task uses hyperparameters such as learning rate and weight decay. These hyperparameters control the behavior of the training algorithm and can have a significant effect on the performance of the resulting model. This part of the notebook show how you can automate tuning these hyperparameters with Vertex Training service.\n",
|
||||
"The training application code for fine-tuning a transformer model for sentiment analysis task uses hyperparameters such as learning rate and weight decay. These hyperparameters control the behavior of the training algorithm and can have a significant effect on the performance of the resulting model. This part of the notebook show how you can automate tuning these hyperparameters with Vertex AI Training service.\n",
|
||||
"\n",
|
||||
"We submit a [Hyperparameter Tuning job](https://cloud.google.com/vertex-ai/docs/training/hyperparameter-tuning-overview) to Vertex Training service by packaging the training application code and dependencies in a Docker container and push the container to Google Container Registry, similar to running a Custom Job on Vertex AI with Custom Container.\n",
|
||||
"We submit a [Hyperparameter Tuning job](https://cloud.google.com/vertex-ai/docs/training/hyperparameter-tuning-overview) to Vertex AI Training service by packaging the training application code and dependencies in a Docker container and push the container to Google Container Registry, similar to running a Custom Job on Vertex AI with Custom Container.\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2163,7 +2163,7 @@
|
||||
"source": [
|
||||
"### How hyperparameter tuning works in Vertex AI?\n",
|
||||
"\n",
|
||||
"Following are the high level steps involved in running a Hyperparameter Tuning job on Vertex Training service:\n",
|
||||
"Following are the high level steps involved in running a Hyperparameter Tuning job on Vertex AI Training service:\n",
|
||||
"\n",
|
||||
"- You define the hyperparameters to tune the model along with the metric (or goal) to optimize\n",
|
||||
"- Vertex AI runs multiple trials of your training application with the hyperparameters and limits you specified - maximum number of trials to run and number of parallel trials. \n",
|
||||
@@ -2297,7 +2297,7 @@
|
||||
"source": [
|
||||
"### Run Hyperparameter Tuning Job on Vertex AI\n",
|
||||
"\n",
|
||||
"Before submitting the hyperparameter tuning job to Vertex AI, push the custom container image with training application to Google Cloud Container Registry and then submit the job to Vertex AI. We will be using the same image used for running Custom Job on Vertex Training service."
|
||||
"Before submitting the hyperparameter tuning job to Vertex AI, push the custom container image with training application to Google Cloud Container Registry and then submit the job to Vertex AI. We will be using the same image used for running Custom Job on Vertex AI Training service."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2326,7 +2326,7 @@
|
||||
"id": "f60fab07d67c"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2346,7 +2346,7 @@
|
||||
"id": "6652aa63ddff"
|
||||
},
|
||||
"source": [
|
||||
"##### **Configure and submit Hyperparameter Tuning Job to Vertex Training service**\n",
|
||||
"##### **Configure and submit Hyperparameter Tuning Job to Vertex AI Training service**\n",
|
||||
"\n",
|
||||
"Configure a [Hyperparameter Tuning Job](https://cloud.google.com/vertex-ai/docs/training/using-hyperparameter-tuning) with the [custom container](https://cloud.google.com/vertex-ai/docs/training/create-custom-container) image with training code and other dependencies.\n",
|
||||
"\n",
|
||||
@@ -2374,7 +2374,7 @@
|
||||
"id": "9d46db3a8b23"
|
||||
},
|
||||
"source": [
|
||||
"Define the training arguments with `hp-tune` argument set to `y` so that training application code can report metrics to Vertex"
|
||||
"Define the training arguments with `hp-tune` argument set to `y` so that training application code can report metrics to Vertex AI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2548,7 +2548,7 @@
|
||||
"\n",
|
||||
"You can monitor the hyperparameter tuning job launched from Cloud Console following the link [here](https://console.cloud.google.com/vertex-ai/training/hyperparameter-tuning-jobs/) or use gcloud CLI command [`gcloud beta ai custom-jobs stream-logs`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/custom-jobs/stream-logs)\n",
|
||||
"\n",
|
||||
""
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2557,7 +2557,7 @@
|
||||
"id": "ba934b434f03"
|
||||
},
|
||||
"source": [
|
||||
"After the job is finished, you can view and format the results of the hyperparameter tuning Trials (run by Vertex Training service) as a Pandas dataframe"
|
||||
"After the job is finished, you can view and format the results of the hyperparameter tuning Trials (run by Vertex AI Training service) as a Pandas dataframe"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2612,7 +2612,7 @@
|
||||
"id": "5dbccb2b7d32"
|
||||
},
|
||||
"source": [
|
||||
"Now from the results of Trials, you can pick the best performing Trial to deploy to Vertex Predictions"
|
||||
"Now from the results of Trials, you can pick the best performing Trial to deploy to Vertex AI Predictions"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2701,8 +2701,8 @@
|
||||
"JOB_NAME=${JOB_PREFIX}-pytorch-hptune-$(date +%Y%m%d%H%M%S)\n",
|
||||
"echo \"Launching hyperparameter tuning job with display name as \"$JOB_NAME\n",
|
||||
"\n",
|
||||
"# BUCKET_NAME: Change to your bucket name\n",
|
||||
"BUCKET_NAME=$1 # <-- CHANGE TO YOUR BUCKET NAME\n",
|
||||
"# BUCKET_NAME is a required parameter to run the cell.\n",
|
||||
"BUCKET_NAME=$1\n",
|
||||
"\n",
|
||||
"# APP_NAME: get application name\n",
|
||||
"APP_NAME=$2\n",
|
||||
@@ -2711,7 +2711,7 @@
|
||||
"JOB_DIR=${BUCKET_NAME}/${JOB_PREFIX}/model/${JOB_NAME}\n",
|
||||
"\n",
|
||||
"# custom container image URI\n",
|
||||
"CUSTOM_TRAIN_IMAGE_URI=f'gcr.io/'${PROJECT_ID}'/pytorch_gpu_train_'${APP_NAME}\n",
|
||||
"CUSTOM_TRAIN_IMAGE_URI='gcr.io/'${PROJECT_ID}'/pytorch_gpu_train_'${APP_NAME}\n",
|
||||
"\n",
|
||||
"# ========================================================\n",
|
||||
"# create hyperparameter tuning configuration file\n",
|
||||
@@ -2772,20 +2772,20 @@
|
||||
"source": [
|
||||
"## Deploying\n",
|
||||
"\n",
|
||||
"Deploying a PyTorch model on [Vertex Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions) requires to use a custom container that serves online predictions. You will deploy a container running [PyTorch's TorchServe](https://pytorch.org/serve/) tool in order to serve predictions from a fine-tuned transformer model from Hugging Face Transformers for sentiment analysis task. You can then use Vertex Predictions to classify sentiment of input texts. \n",
|
||||
"Deploying a PyTorch model on [Vertex AI Predictions](https://cloud.google.com/vertex-ai/docs/predictions/getting-predictions) requires to use a custom container that serves online predictions. You will deploy a container running [PyTorch's TorchServe](https://pytorch.org/serve/) tool in order to serve predictions from a fine-tuned transformer model from Hugging Face Transformers for sentiment analysis task. You can then use Vertex AI Predictions to classify sentiment of input texts. \n",
|
||||
"\n",
|
||||
"### Deploying model on Vertex Predictions with custom container\n",
|
||||
"### Deploying model on Vertex AI Predictions with custom container\n",
|
||||
"\n",
|
||||
"To use a custom container to serve predictions from a PyTorch model, you must provide Vertex AI with a Docker container image that runs an HTTP server, such as TorchServe in this case. Please refer to [documentation](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) that describes the container image requirements to be compatible with Vertex Predictions.\n",
|
||||
"To use a custom container to serve predictions from a PyTorch model, you must provide Vertex AI with a Docker container image that runs an HTTP server, such as TorchServe in this case. Please refer to [documentation](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) that describes the container image requirements to be compatible with Vertex AI Predictions.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Essentially, to deploy a PyTorch model on Vertex Predictions following are the steps:\n",
|
||||
"Essentially, to deploy a PyTorch model on Vertex AI Predictions following are the steps:\n",
|
||||
"\n",
|
||||
"1. Package the trained model artifacts including [default](https://pytorch.org/serve/#default-handlers) or [custom](https://pytorch.org/serve/custom_service.html) handlers by creating an archive file using [Torch model archiver](https://github.com/pytorch/serve/tree/master/model-archiver)\n",
|
||||
"2. Build a [custom container](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) compatible with Vertex Predictions to serve the model using Torchserve\n",
|
||||
"3. Upload the model with custom container image to serve predictions as a Vertex Model resource\n",
|
||||
"4. Create a Vertex Endpoint and [deploy the model](https://cloud.google.com/vertex-ai/docs/predictions/deploy-model-api) resource"
|
||||
"2. Build a [custom container](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements) compatible with Vertex AI Predictions to serve the model using Torchserve\n",
|
||||
"3. Upload the model with custom container image to serve predictions as a Vertex AI Model resource\n",
|
||||
"4. Create a Vertex AI Endpoint and [deploy the model](https://cloud.google.com/vertex-ai/docs/predictions/deploy-model-api) resource"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2815,7 +2815,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile predictor/custom_text_handler.py\n",
|
||||
"%%writefile predictor/custom_handler.py\n",
|
||||
"\n",
|
||||
"import os\n",
|
||||
"import json\n",
|
||||
@@ -2870,7 +2870,8 @@
|
||||
" with open(mapping_file_path) as f:\n",
|
||||
" self.mapping = json.load(f)\n",
|
||||
" else:\n",
|
||||
" logger.warning('Missing the index_to_name.json file. Inference output will not include class name.')\n",
|
||||
" logger.warning('Missing the index_to_name.json file. Inference output will default.')\n",
|
||||
" self.mapping = {\"0\": \"Negative\", \"1\": \"Positive\"}\n",
|
||||
"\n",
|
||||
" self.initialized = True\n",
|
||||
"\n",
|
||||
@@ -3047,10 +3048,13 @@
|
||||
"FROM pytorch/torchserve:latest-cpu\n",
|
||||
"\n",
|
||||
"# install dependencies\n",
|
||||
"RUN python3 -m pip install --upgrade pip\n",
|
||||
"RUN pip3 install transformers\n",
|
||||
"\n",
|
||||
"USER model-server\n",
|
||||
"\n",
|
||||
"# copy model artifacts, custom handler and other dependencies\n",
|
||||
"COPY ./custom_text_handler.py /home/model-server/\n",
|
||||
"COPY ./custom_handler.py /home/model-server/\n",
|
||||
"COPY ./index_to_name.json /home/model-server/\n",
|
||||
"COPY ./model/$APP_NAME/ /home/model-server/\n",
|
||||
"\n",
|
||||
@@ -3070,7 +3074,7 @@
|
||||
" --model-name=$APP_NAME \\\n",
|
||||
" --version=1.0 \\\n",
|
||||
" --serialized-file=/home/model-server/pytorch_model.bin \\\n",
|
||||
" --handler=/home/model-server/custom_text_handler.py \\\n",
|
||||
" --handler=/home/model-server/custom_handler.py \\\n",
|
||||
" --extra-files \"/home/model-server/config.json,/home/model-server/tokenizer.json,/home/model-server/training_args.bin,/home/model-server/tokenizer_config.json,/home/model-server/special_tokens_map.json,/home/model-server/vocab.txt,/home/model-server/index_to_name.json\" \\\n",
|
||||
" --export-path=/home/model-server/model-store\n",
|
||||
"\n",
|
||||
@@ -3129,7 +3133,7 @@
|
||||
"source": [
|
||||
"#### **Run the container locally** ***[Optional]***\n",
|
||||
"\n",
|
||||
"Before push the container image to Container Registry to use it with Vertex Predictions, you can run it as a container in your local environment to verify that the server works as expected"
|
||||
"Before push the container image to Container Registry to use it with Vertex AI Predictions, you can run it as a container in your local environment to verify that the server works as expected"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3267,9 +3271,9 @@
|
||||
"id": "69477b3a00c0"
|
||||
},
|
||||
"source": [
|
||||
"#### **Deploying the serving container to Vertex Predictions**\n",
|
||||
"#### **Deploying the serving container to Vertex AI Predictions**\n",
|
||||
"\n",
|
||||
"We create a model resource on Vertex AI and deploy the model to a Vertex Endpoints. You must deploy a model to an endpoint before using the model. The deployed model runs the custom container image to serve predictions. "
|
||||
"We create a model resource on Vertex AI and deploy the model to a Vertex AI Endpoints. You must deploy a model to an endpoint before using the model. The deployed model runs the custom container image to serve predictions. "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3300,7 +3304,7 @@
|
||||
"id": "a3da91e19af4"
|
||||
},
|
||||
"source": [
|
||||
"##### **Initialize the Vertex SDK for Python**"
|
||||
"##### **Initialize the Vertex AI SDK for Python**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3437,7 +3441,7 @@
|
||||
"id": "bc4673478269"
|
||||
},
|
||||
"source": [
|
||||
"#### **Invoking the Endpoint with deployed Model using Vertex SDK to make predictions**"
|
||||
"#### **Invoking the Endpoint with deployed Model using Vertex AI SDK to make predictions**"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3487,7 +3491,7 @@
|
||||
"source": [
|
||||
"##### **Formatting input for online prediction**\n",
|
||||
"\n",
|
||||
"For online prediction requests, the prediction input instances must be formatted as JSON with base64 encoding as shown here:\n",
|
||||
"This notebook uses [Torchserve's KServe based inference API](https://pytorch.org/serve/inference_api.html#kserve-inference-api) which is also [Vertex AI Predictions compatible format](https://cloud.google.com/vertex-ai/docs/predictions/custom-container-requirements#prediction). For online prediction requests, format the prediction input instances as JSON with base64 encoding as shown here:\n",
|
||||
"\n",
|
||||
"```\n",
|
||||
"[\n",
|
||||
@@ -3560,9 +3564,9 @@
|
||||
},
|
||||
"source": [
|
||||
"##### ***[Optional]*** **Make prediction requests using gcloud CLI**\n",
|
||||
"You can also call the Vertex Endpoint to make predictions using [`gcloud beta ai endpoints predict`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/endpoints/predict). \n",
|
||||
"You can also call the Vertex AI Endpoint to make predictions using [`gcloud beta ai endpoints predict`](https://cloud.google.com/sdk/gcloud/reference/beta/ai/endpoints/predict). \n",
|
||||
"\n",
|
||||
"The following cell shows how to make a prediction request to Vertex Endpoints using `gcloud` CLI: "
|
||||
"The following cell shows how to make a prediction request to Vertex AI Endpoints using `gcloud` CLI: "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3653,12 +3657,12 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_custom_job = True\n",
|
||||
"delete_hp_tuning_job = True\n",
|
||||
"delete_custom_job = False\n",
|
||||
"delete_hp_tuning_job = False\n",
|
||||
"delete_endpoint = True\n",
|
||||
"delete_model = True\n",
|
||||
"delete_bucket = True\n",
|
||||
"delete_image = True"
|
||||
"delete_model = False\n",
|
||||
"delete_bucket = False\n",
|
||||
"delete_image = False"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -3686,7 +3690,7 @@
|
||||
"\n",
|
||||
"client_options = {\"api_endpoint\": API_ENDPOINT}\n",
|
||||
"\n",
|
||||
"# Initialize Vertex SDK\n",
|
||||
"# Initialize Vertex AI SDK\n",
|
||||
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
]
|
||||
},
|
||||
@@ -3924,7 +3928,7 @@
|
||||
"if delete_bucket and \"BUCKET_NAME\" in globals():\n",
|
||||
" print(f\"Deleting all contents from the bucket {BUCKET_NAME}\")\n",
|
||||
"\n",
|
||||
" shell_output=! gsutil du -as $BUCKET_NAME\n",
|
||||
" shell_output = ! gsutil du -as $BUCKET_NAME\n",
|
||||
" print(\n",
|
||||
" f\"Size of the bucket {BUCKET_NAME} before deleting = {shell_output[0].split()[0]} bytes\"\n",
|
||||
" )\n",
|
||||
@@ -3932,7 +3936,7 @@
|
||||
" # uncomment below line to delete contents of the bucket\n",
|
||||
" # ! gsutil rm -r $BUCKET_NAME\n",
|
||||
"\n",
|
||||
" shell_output=! gsutil du -as $BUCKET_NAME\n",
|
||||
" shell_output = ! gsutil du -as $BUCKET_NAME\n",
|
||||
" if float(shell_output[0].split()[0]) > 0:\n",
|
||||
" print(\n",
|
||||
" \"PLEASE UNCOMMENT LINE TO DELETE BUCKET. CONTENT FROM THE BUCKET NOT DELETED\"\n",
|
||||
|
||||
@@ -188,11 +188,14 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! pip3 install {USER_FLAG} google-cloud-aiplatform==1.0.1\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-pipeline-components==0.1.3\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-aiplatform\n",
|
||||
"! pip3 install {USER_FLAG} google-cloud-pipeline-components\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade kfp\n",
|
||||
"! pip3 install {USER_FLAG} numpy==1.20.3\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tensorflow"
|
||||
"! pip3 install {USER_FLAG} numpy\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tensorflow\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade pillow\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade tf-agents\n",
|
||||
"! pip3 install {USER_FLAG} --upgrade fastapi"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -287,7 +290,7 @@
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
@@ -518,6 +521,7 @@
|
||||
"import os\n",
|
||||
"import sys\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"from kfp.v2 import compiler, dsl\n",
|
||||
"from kfp.v2.google.client import AIPlatformClient"
|
||||
@@ -561,13 +565,34 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H3530hdGGilo"
|
||||
"id": "895ac243c125"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Dataset parameters\n",
|
||||
"RAW_DATA_PATH = \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" # Location of the MovieLens 100K dataset's \"u.data\" file.\n",
|
||||
"\n",
|
||||
"RAW_DATA_PATH = \"gs://[your-bucket-name]/raw_data/u.data\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "62bfb9a820f6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download the sample data into your RAW_DATA_PATH\n",
|
||||
"! gsutil cp \"gs://cloud-samples-data/vertex-ai/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/u.data\" $RAW_DATA_PATH"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H3530hdGGilo"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Pipeline parameters\n",
|
||||
"PIPELINE_NAME = \"movielens-pipeline\" # Pipeline display name.\n",
|
||||
"ENABLE_CACHING = False # Whether to enable execution caching for the pipeline.\n",
|
||||
@@ -635,7 +660,7 @@
|
||||
"source": [
|
||||
"#### Run unit tests on the Generator component\n",
|
||||
"\n",
|
||||
"Before running the command, fill in `RAW_DATA_PATH` in [`src/generator/test_generator_component.py`](src/generator/test_generator_component.py)."
|
||||
"Before running the command, you should update the `RAW_DATA_PATH` in [`src/generator/test_generator_component.py`](src/generator/test_generator_component.py)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -713,12 +738,12 @@
|
||||
"TRAINING_ARTIFACTS_DIR = (\n",
|
||||
" f\"{BUCKET_NAME}/artifacts\" # Root directory for training artifacts.\n",
|
||||
")\n",
|
||||
"TRAINING_REPLICA_COUNT = \"1\" # Number of replica to run the custom training job.\n",
|
||||
"TRAINING_REPLICA_COUNT = 1 # Number of replica to run the custom training job.\n",
|
||||
"TRAINING_MACHINE_TYPE = (\n",
|
||||
" \"n1-standard-4\" # Type of machine to run the custom training job.\n",
|
||||
")\n",
|
||||
"TRAINING_ACCELERATOR_TYPE = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Type of accelerators to run the custom training job.\n",
|
||||
"TRAINING_ACCELERATOR_COUNT = \"0\" # Number of accelerators for the custom training job."
|
||||
"TRAINING_ACCELERATOR_COUNT = 0 # Number of accelerators for the custom training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -769,8 +794,12 @@
|
||||
"TRAINED_POLICY_DISPLAY_NAME = (\n",
|
||||
" \"movielens-trained-policy\" # Display name of the uploaded and deployed policy.\n",
|
||||
")\n",
|
||||
"TRAFFIC_SPLIT = {\"0\": 100}\n",
|
||||
"ENDPOINT_DISPLAY_NAME = \"movielens-endpoint\" # Display name of the prediction endpoint.\n",
|
||||
"ENDPOINT_MACHINE_TYPE = \"n1-standard-4\" # Type of machine of the prediction endpoint."
|
||||
"ENDPOINT_MACHINE_TYPE = \"n1-standard-4\" # Type of machine of the prediction endpoint.\n",
|
||||
"ENDPOINT_REPLICA_COUNT = 1 # Number of replicas of the prediction endpoint.\n",
|
||||
"ENDPOINT_ACCELERATOR_TYPE = \"ACCELERATOR_TYPE_UNSPECIFIED\" # Type of accelerators to run the custom training job.\n",
|
||||
"ENDPOINT_ACCELERATOR_COUNT = 0 # Number of accelerators for the custom training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -900,16 +929,17 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components.experimental.custom_job import utils\n",
|
||||
"from kfp.components import load_component_from_url\n",
|
||||
"\n",
|
||||
"generate_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/generator/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/generator/component.yaml\"\n",
|
||||
")\n",
|
||||
"ingest_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
")\n",
|
||||
"train_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -978,7 +1008,7 @@
|
||||
" bigquery_location=bigquery_location,\n",
|
||||
" bigquery_table_id=bigquery_table_id,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" \n",
|
||||
" # Run the Ingester component.\n",
|
||||
" ingest_task = ingest_op(\n",
|
||||
" project_id=project_id,\n",
|
||||
@@ -988,7 +1018,16 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Run the Trainer component and submit custom job to Vertex AI.\n",
|
||||
" train_task = train_op(\n",
|
||||
" # Convert the train_op component into a Vertex AI Custom Job pre-built component\n",
|
||||
" custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" component_spec=train_op,\n",
|
||||
" replica_count=TRAINING_REPLICA_COUNT,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train_task = custom_job_training_op(\n",
|
||||
" training_artifacts_dir=training_artifacts_dir,\n",
|
||||
" tfrecord_file=ingest_task.outputs[\"tfrecord_file\"],\n",
|
||||
" num_epochs=num_epochs,\n",
|
||||
@@ -996,28 +1035,10 @@
|
||||
" num_actions=num_actions,\n",
|
||||
" tikhonov_weight=tikhonov_weight,\n",
|
||||
" agent_alpha=agent_alpha,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" worker_pool_specs = [\n",
|
||||
" {\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": train_task.container.image,\n",
|
||||
" },\n",
|
||||
" \"replicaCount\": TRAINING_REPLICA_COUNT,\n",
|
||||
" \"machineSpec\": {\n",
|
||||
" \"machineType\": TRAINING_MACHINE_TYPE,\n",
|
||||
" \"acceleratorType\": TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" \"acceleratorCount\": TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" train_task.custom_job_spec = {\n",
|
||||
" \"displayName\": train_task.name,\n",
|
||||
" \"jobSpec\": {\n",
|
||||
" \"workerPoolSpecs\": worker_pool_specs,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Run the Deployer components.\n",
|
||||
" # Upload the trained policy as a model.\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
@@ -1034,11 +1055,14 @@
|
||||
" # Deploy the uploaded, trained policy to the created endpoint. (This operation\n",
|
||||
" # has to occur after both model uploading and endpoint creation complete.)\n",
|
||||
" gcc_aip.ModelDeployOp(\n",
|
||||
" project=project_id,\n",
|
||||
" endpoint=endpoint_create_op.outputs[\"endpoint\"],\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" deployed_model_display_name=TRAINED_POLICY_DISPLAY_NAME,\n",
|
||||
" machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" traffic_split=TRAFFIC_SPLIT,\n",
|
||||
" dedicated_resources_machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_accelerator_type=ENDPOINT_ACCELERATOR_TYPE,\n",
|
||||
" dedicated_resources_accelerator_count=ENDPOINT_ACCELERATOR_COUNT,\n",
|
||||
" dedicated_resources_min_replica_count=ENDPOINT_REPLICA_COUNT,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
@@ -1053,12 +1077,11 @@
|
||||
"# Compile the authored pipeline.\n",
|
||||
"compiler.Compiler().compile(pipeline_func=pipeline, package_path=PIPELINE_SPEC_PATH)\n",
|
||||
"\n",
|
||||
"# Createa Vertex AI client.\n",
|
||||
"api_client = AIPlatformClient(project_id=PROJECT_ID, region=REGION)\n",
|
||||
"\n",
|
||||
"# Create a pipeline run job.\n",
|
||||
"response = api_client.create_run_from_job_spec(\n",
|
||||
" job_spec_path=PIPELINE_SPEC_PATH,\n",
|
||||
"job = aiplatform.PipelineJob(\n",
|
||||
" display_name=f\"{PIPELINE_NAME}-startup\",\n",
|
||||
" template_path=PIPELINE_SPEC_PATH,\n",
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" # Pipeline configs\n",
|
||||
" \"project_id\": PROJECT_ID,\n",
|
||||
@@ -1070,7 +1093,9 @@
|
||||
" \"bigquery_table_id\": BIGQUERY_TABLE_ID,\n",
|
||||
" },\n",
|
||||
" enable_caching=ENABLE_CACHING,\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"job.run()"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1111,7 +1136,11 @@
|
||||
"SIMULATOR_SCHEDULE = \"*/5 * * * *\" # Cloud Scheduler cron job schedule for the Simulator. Eg. \"*/5 * * * *\" means every 5 mins.\n",
|
||||
"SIMULATOR_SCHEDULER_MESSAGE = (\n",
|
||||
" \"simulator-message\" # Cloud Scheduler message for the Simulator.\n",
|
||||
")"
|
||||
")\n",
|
||||
"# TF-Agents RL configs\n",
|
||||
"BATCH_SIZE = 8\n",
|
||||
"RANK_K = 20\n",
|
||||
"NUM_ACTIONS = 20"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1221,7 +1250,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoints = ! gcloud beta ai endpoints list \\\n",
|
||||
"endpoints = ! gcloud ai endpoints list \\\n",
|
||||
" --region=$REGION \\\n",
|
||||
" --filter=display_name=$ENDPOINT_DISPLAY_NAME\n",
|
||||
"print(\"\\n\".join(endpoints), \"\\n\")\n",
|
||||
@@ -1424,13 +1453,11 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from kfp.components import load_component_from_url\n",
|
||||
"\n",
|
||||
"ingest_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/ingester/component.yaml\"\n",
|
||||
")\n",
|
||||
"train_op = load_component_from_url(\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/68d6cf46ee22a9b9295d62ea71996150baf8db94/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
" \"https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/62a2a7611499490b4b04d731d48a7ba87c2d636f/community-content/tf_agents_bandits_movie_recommendation_with_kfp_and_vertex_sdk/mlops_pipeline_tf_agents_bandits_movie_recommendation/src/trainer/component.yaml\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -1481,7 +1508,16 @@
|
||||
" )\n",
|
||||
"\n",
|
||||
" # Run the Trainer component and submit custom job to Vertex AI.\n",
|
||||
" train_task = train_op(\n",
|
||||
" # Convert the train_op component into a Vertex AI Custom Job pre-built component\n",
|
||||
" custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" component_spec=train_op,\n",
|
||||
" replica_count=TRAINING_REPLICA_COUNT,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" train_task = custom_job_training_op(\n",
|
||||
" training_artifacts_dir=training_artifacts_dir,\n",
|
||||
" tfrecord_file=ingest_task.outputs[\"tfrecord_file\"],\n",
|
||||
" num_epochs=num_epochs,\n",
|
||||
@@ -1489,28 +1525,10 @@
|
||||
" num_actions=num_actions,\n",
|
||||
" tikhonov_weight=tikhonov_weight,\n",
|
||||
" agent_alpha=agent_alpha,\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" worker_pool_specs = [\n",
|
||||
" {\n",
|
||||
" \"containerSpec\": {\n",
|
||||
" \"imageUri\": train_task.container.image,\n",
|
||||
" },\n",
|
||||
" \"replicaCount\": TRAINING_REPLICA_COUNT,\n",
|
||||
" \"machineSpec\": {\n",
|
||||
" \"machineType\": TRAINING_MACHINE_TYPE,\n",
|
||||
" \"acceleratorType\": TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" \"acceleratorCount\": TRAINING_ACCELERATOR_COUNT,\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
" ]\n",
|
||||
" train_task.custom_job_spec = {\n",
|
||||
" \"displayName\": train_task.name,\n",
|
||||
" \"jobSpec\": {\n",
|
||||
" \"workerPoolSpecs\": worker_pool_specs,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Run the Deployer components.\n",
|
||||
" # Upload the trained policy as a model.\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
@@ -1527,11 +1545,13 @@
|
||||
" # Deploy the uploaded, trained policy to the created endpoint. (This operation\n",
|
||||
" # has to occur after both model uploading and endpoint creation complete.)\n",
|
||||
" gcc_aip.ModelDeployOp(\n",
|
||||
" project=project_id,\n",
|
||||
" endpoint=endpoint_create_op.outputs[\"endpoint\"],\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" deployed_model_display_name=TRAINED_POLICY_DISPLAY_NAME,\n",
|
||||
" machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_machine_type=ENDPOINT_MACHINE_TYPE,\n",
|
||||
" dedicated_resources_accelerator_type=ENDPOINT_ACCELERATOR_TYPE,\n",
|
||||
" dedicated_resources_accelerator_count=ENDPOINT_ACCELERATOR_COUNT,\n",
|
||||
" dedicated_resources_min_replica_count=ENDPOINT_REPLICA_COUNT,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -39,14 +39,15 @@ outputs:
|
||||
- {name: bigquery_table_id, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- (PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'google-cloud-bigquery==2.20.0' 'tensorflow==2.5.0' 'tf-agents==0.8.0' || PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||
python3 -m pip install --quiet --no-warn-script-location 'google-cloud-bigquery==2.20.0'
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' --user) && "$0" "$@"
|
||||
'google-cloud-bigquery==2.20.0' 'pillow' 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
|| PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'google-cloud-bigquery==2.20.0' 'pillow' 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
--user) && "$0" "$@"
|
||||
- sh
|
||||
- -ec
|
||||
- |
|
||||
@@ -296,7 +297,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -20,7 +20,7 @@ outputs:
|
||||
- {name: tfrecord_file, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
@@ -187,7 +187,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
google-cloud-bigquery==2.20.0
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
|
||||
@@ -1,2 +1,4 @@
|
||||
google-cloud-pubsub==2.5.0
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
dataclasses==0.6
|
||||
google-cloud-aiplatform==1.8.1
|
||||
tensorflow==2.5.3
|
||||
pillow==9.0.1
|
||||
tf-agents==0.8.0
|
||||
@@ -27,14 +27,14 @@ outputs:
|
||||
- {name: training_artifacts_dir, type: String}
|
||||
implementation:
|
||||
container:
|
||||
image: tensorflow/tensorflow:2.5.0
|
||||
image: python:3.7
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- (PIP_DISABLE_PIP_VERSION_CHECK=1 python3 -m pip install --quiet --no-warn-script-location
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' || PIP_DISABLE_PIP_VERSION_CHECK=1 python3
|
||||
-m pip install --quiet --no-warn-script-location 'tensorflow==2.5.0' 'tf-agents==0.8.0'
|
||||
--user) && "$0" "$@"
|
||||
'tensorflow==2.5.0' 'tf-agents==0.8.0' 'Pillow' || PIP_DISABLE_PIP_VERSION_CHECK=1
|
||||
python3 -m pip install --quiet --no-warn-script-location 'tensorflow==2.5.0'
|
||||
'tf-agents==0.8.0' 'Pillow' --user) && "$0" "$@"
|
||||
- sh
|
||||
- -ec
|
||||
- |
|
||||
@@ -270,7 +270,8 @@ implementation:
|
||||
|
||||
def _serialize_str(str_value: str) -> str:
|
||||
if not isinstance(str_value, str):
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(str(str_value), str(type(str_value))))
|
||||
raise TypeError('Value "{}" has type "{}" instead of str.'.format(
|
||||
str(str_value), str(type(str_value))))
|
||||
return str_value
|
||||
|
||||
import argparse
|
||||
|
||||
@@ -22,13 +22,13 @@ from src.training import task
|
||||
|
||||
|
||||
# Paths and configurations
|
||||
DATA_PATH = "gs://[your-bucket-name]/[your-dataset-dir]/u.data" # FILL IN
|
||||
DATA_PATH = "gs://[your-bucket-name]/artifacts/u.data" # FILL IN
|
||||
ROOT_DIR = "gs://[your-bucket-name]/artifacts" # FILL IN
|
||||
ARTIFACTS_DIR = "gs://[your-bucket-name]/artifacts" # FILL IN
|
||||
PROFILER_DIR = "gs://[your-bucket-name]/profiler" # FILL IN
|
||||
HPTUNING_RESULT_DIR = "[your-hptuning-result-dir]/" # FILL IN
|
||||
HPTUNING_RESULT_PATH = os.path.join(HPTUNING_RESULT_DIR,
|
||||
"[your-file-name].json") # FILL IN
|
||||
"result.json") # FILL IN
|
||||
RAW_BUCKET_NAME = "[your-hptuning-result-bucket-name]" # FILL IN
|
||||
|
||||
# Hyperparameters
|
||||
|
||||
@@ -1 +1 @@
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
@@ -1 +1 @@
|
||||
tensorflow==2.5.2
|
||||
tensorflow==2.5.3
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -113,8 +113,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud beta ai custom-jobs local-run \\\n",
|
||||
" --base-image=$BASE_IMAGE_URI \\\n",
|
||||
"! gcloud ai custom-jobs local-run \\\n",
|
||||
" --executor-image-uri=$BASE_IMAGE_URI \\\n",
|
||||
" --script=$SCRIPT_PATH \\\n",
|
||||
" --output-image-uri=$OUTPUT_IMAGE_NAME \\\n",
|
||||
" -- \\\n",
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
The [official](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/official) folder contains notebooks organized by Google Cloud product.
|
||||
|
||||
The [community](https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/main/notebooks/community) folder contains notebooks that aren't officially supported by Google.
|
||||
|
||||
Contributions to the repo should use the [notebook template](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb) as a starting point.
|
||||
@@ -3,16 +3,19 @@
|
||||
# @global-owner1 and @global-owner2 will be requested for
|
||||
# review when someone opens a pull request.
|
||||
|
||||
/sdk/sdk_* @aferlitsch
|
||||
/gapic @aferlitsch
|
||||
/ml_ops @aferlitsch
|
||||
/model_monitoring/* @mco
|
||||
/sdk/sdk_* @andrewferlitsch
|
||||
/gapic @andrewferlitsch
|
||||
/ml_ops @andrewferlitsch
|
||||
/model_monitoring/* @mco-gh
|
||||
/structured_data/rapid_prototyping_* @rafael-carvalho
|
||||
|
||||
/managed_notebooks/ @notebooks-team
|
||||
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
|
||||
/managed_notebooks/
|
||||
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
|
||||
/pipelines/google_cloud_pipeline_components_TPU_model_train_upload_deploy.ipynb @brianchunkang
|
||||
/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb @thehardikv
|
||||
/sdk/sdk_automl_forecasting_evaluating_a_model.ipynb @thehardikv
|
||||
/matching_engine @yinghsienwu
|
||||
/neo4j @benofben @htappen
|
||||
/matching_engine/sdk_matching_engine_for_indexing.ipynb @ivanmkc
|
||||
/matching_engine/matching_engine_for_indexing.ipynb @yinghsienwu
|
||||
/sdk/pytorch_lightning_custom_container_training.ipynb @brianchunkang
|
||||
/tensorboard @yfang1
|
||||
/feature_store @nayaknishant @morgandu
|
||||
/vertex_endpoints/tf_hub_obj_detection/deploy_tfhub_object_detection_on_vertex_endpoints.ipynb @entrpn
|
||||
/vertex_endpoints/nvidia-triton/nvidia-triton-custom-container-prediction.ipynb @RajeshThallam
|
||||
|
||||
|
After Width: | Height: | Size: 153 KiB |
|
After Width: | Height: | Size: 138 KiB |
|
After Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 182 KiB |
|
After Width: | Height: | Size: 142 KiB |
|
After Width: | Height: | Size: 382 KiB |
|
After Width: | Height: | Size: 445 KiB |
|
After Width: | Height: | Size: 63 KiB |
|
After Width: | Height: | Size: 59 KiB |
@@ -0,0 +1,823 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d1cc1c1fa076"
|
||||
},
|
||||
"source": [
|
||||
"# Pricing Optimization \n",
|
||||
"## Table of contents\n",
|
||||
"* [Overview](#section-1)\n",
|
||||
"* [Dataset](#section-2)\n",
|
||||
"* [Objective](#section-3)\n",
|
||||
"* [Costs](#section-4)\n",
|
||||
"* [Create a BigQuery dataset](#section-5)\n",
|
||||
"* [Load the dataset from Cloud Storage](#section-6)\n",
|
||||
"* [Data analysis](#section-7)\n",
|
||||
"* [Preprocess the data for training](#section-8)\n",
|
||||
"* [Train the model using BigQuery ML](#section-9)\n",
|
||||
"* [Generate forecasts from the model](#section-10)\n",
|
||||
"* [Interpret the results to choose the best price](#section-11)\n",
|
||||
"* [Clean up](#section-12)\n",
|
||||
"\n",
|
||||
"## Overview\n",
|
||||
"<a name=\"section-1\"></a>\n",
|
||||
"\n",
|
||||
"This notebook demonstrates analysis of pricing optimization on [CDM Pricing Data](https://github.com/trifacta/trifacta-google-cloud/tree/main/design-pattern-pricing-optimization) and automating the workflow using Vertex AI Workbench managed notebooks.\n",
|
||||
"\n",
|
||||
"*Note: This notebook file was developed to run in a [Vertex AI Workbench managed notebooks](https://console.cloud.google.com/vertex-ai/workbench/list/managed) instance using the Python (Local) kernel. Some components of this notebook may not work in other notebook environments.*\n",
|
||||
"\n",
|
||||
"## Dataset\n",
|
||||
"<a name=\"section-2\"></a>\n",
|
||||
"\n",
|
||||
"The dataset used in this notebook is a part of the [CDM Pricing dataset](https://github.com/trifacta/trifacta-google-cloud/blob/main/design-pattern-pricing-optimization/CDM_Pricing_large_table.csv), which consists of product sales information on specified dates.\n",
|
||||
"\n",
|
||||
"## Objective\n",
|
||||
"<a name=\"section-3\"></a>\n",
|
||||
"\n",
|
||||
"The objective of this notebook is to build a pricing optimization model using Vertex AI. The following steps have been followed: \n",
|
||||
"\n",
|
||||
"- Load the required dataset from a Cloud Storage bucket.\n",
|
||||
"- Analyze the fields present in the dataset.\n",
|
||||
"- Process the data to build a model.\n",
|
||||
"- Build a BigQuery ML forecast model on the processed data.\n",
|
||||
"- Get forecasted values from the BigQuery ML model.\n",
|
||||
"- Interpret the forecasts to identify the best prices.\n",
|
||||
"- Clean up.\n",
|
||||
"\n",
|
||||
"## Costs\n",
|
||||
"<a name=\"section-4\"></a>\n",
|
||||
"\n",
|
||||
"This tutorial uses the following billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"- Vertex AI\n",
|
||||
"- BigQuery\n",
|
||||
"- Cloud Storage\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI\n",
|
||||
"pricing](https://cloud.google.com/vertex-ai/pricing), [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and [Cloud Storage\n",
|
||||
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
|
||||
"Calculator](https://cloud.google.com/products/calculator/)\n",
|
||||
"to generate a cost estimate based on your projected usage.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5ed1f5e85640"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "c3f30148b66d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"PROJECT_ID = \"\"\n",
|
||||
"\n",
|
||||
"# Get your Google Cloud project ID from gcloud\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID: \", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "750bf2883c2d"
|
||||
},
|
||||
"source": [
|
||||
"Otherwise, set your project ID here."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "3c6db1ca88b9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
|
||||
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2a1c270c7d34"
|
||||
},
|
||||
"source": [
|
||||
"### Import the required libraries and define constants\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "acc6fac1fa55"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"import pandas as pd\n",
|
||||
"import seaborn as sns\n",
|
||||
"from google.cloud import bigquery\n",
|
||||
"from google.cloud.bigquery import Client"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a06006dff8f9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DATASET = \"[your-bigquery-dataset-id]\" # set the BigQuery dataset-id\n",
|
||||
"TRAINING_DATA_TABLE = \"[your-bigquery-table-id-to-store-the-training-data]\" # set the BigQuery table-id to store the training data"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "016c3d47cc69"
|
||||
},
|
||||
"source": [
|
||||
"## Create a BigQuery dataset\n",
|
||||
"<a name=\"section-5\"></a>\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "12ccd8d7956e"
|
||||
},
|
||||
"source": [
|
||||
"#@bigquery\n",
|
||||
"-- create a dataset in BigQuery\n",
|
||||
"\n",
|
||||
"CREATE SCHEMA pricing_optimization\n",
|
||||
"OPTIONS(\n",
|
||||
" location=\"us\"\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "c106b978a79b"
|
||||
},
|
||||
"source": [
|
||||
"## Load the dataset from Cloud Storage\n",
|
||||
"<a name=\"section-6\"></a>\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "8aeae9da9796"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DATA_LOCATION = \"gs://cloud-samples-data/ai-platform-unified/datasets/tabular/cdm_pricing_large_table.csv\"\n",
|
||||
"df = pd.read_csv(DATA_LOCATION)\n",
|
||||
"print(df.shape)\n",
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "7b98d5f09842"
|
||||
},
|
||||
"source": [
|
||||
"You will build a forecast model on this data and thus determine the best price for a product. For this type of model, you will not be using many fields: only the sales and price related ones. For the current execrcise, focus on the following fields:\n",
|
||||
"\n",
|
||||
"- `Product_ID`\n",
|
||||
"- `Customer_Hierarchy`\n",
|
||||
"- `Fiscal_Date`\n",
|
||||
"- `List_Price_Converged`\n",
|
||||
"- `Invoiced_quantity_in_Pieces`\n",
|
||||
"- `Net_Sales`\n",
|
||||
"\n",
|
||||
"## Data Analysis\n",
|
||||
"<a name=\"section-7\"></a>\n",
|
||||
"\n",
|
||||
"First, explore the data and distributions.\n",
|
||||
"\n",
|
||||
"Select the required columns from the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "af4b41c5eb1f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"id_col = \"Product_ID\"\n",
|
||||
"date_col = \"Fiscal_Date\"\n",
|
||||
"categ_cols = [\"Customer_Hierarchy\"]\n",
|
||||
"num_cols = [\"List_Price_Converged\", \"Invoiced_quantity_in_Pieces\", \"Net_Sales\"]\n",
|
||||
"\n",
|
||||
"df = df[[id_col, date_col] + categ_cols + num_cols].copy()\n",
|
||||
"df.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3d780043ee5b"
|
||||
},
|
||||
"source": [
|
||||
"Check the column types and null values in the dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f54c445a1288"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df.info()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cd817b414c4d"
|
||||
},
|
||||
"source": [
|
||||
"This data description reveals that there are no null values in the data. Also, the field `Fiscal_Date` which is a date field is loaded as an object type. \n",
|
||||
"\n",
|
||||
"Change the type of the date field to datetime."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b160fac085c8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df[\"Fiscal_Date\"] = pd.to_datetime(df[\"Fiscal_Date\"], infer_datetime_format=True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fb4778578064"
|
||||
},
|
||||
"source": [
|
||||
"Plot the distributions for the categorical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "dd0467cd57c3"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in categ_cols:\n",
|
||||
" df[i].value_counts(normalize=True).plot(kind=\"bar\")\n",
|
||||
" plt.title(i)\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "145deed255e0"
|
||||
},
|
||||
"source": [
|
||||
"Plot the distributions for the numerical fields."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f934137c6d82"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in num_cols:\n",
|
||||
" _, ax = plt.subplots(1, 2, figsize=(10, 4))\n",
|
||||
" df[i].plot(kind=\"box\", ax=ax[0])\n",
|
||||
" df[i].plot(kind=\"hist\", ax=ax[1])\n",
|
||||
" ax[0].set_title(i + \"-Boxplot\")\n",
|
||||
" ax[1].set_title(i + \"-Histogram\")\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f9b9c2e58380"
|
||||
},
|
||||
"source": [
|
||||
"Check the maximum date and minimum date in Fiscal_Date column."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2a10aa689f9d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"print(df[\"Fiscal_Date\"].max())\n",
|
||||
"print(df[\"Fiscal_Date\"].min())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "4834f63e2e59"
|
||||
},
|
||||
"source": [
|
||||
"Check the product distribution across each category."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4664877f5304"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"grp_cols = [\"Customer_Hierarchy\", \"Product_ID\"]\n",
|
||||
"grp_df = df[grp_cols].groupby(by=grp_cols).count().reset_index()\n",
|
||||
"grp_df.groupby(\"Customer_Hierarchy\").nunique()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "01ed02b9c8fd"
|
||||
},
|
||||
"source": [
|
||||
"Check the percentage changes in the orders based on the percentage changes in the price."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "0b2c428cb135"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# aggregate the data\n",
|
||||
"df_aggr = (\n",
|
||||
" df.groupby([\"Product_ID\", \"List_Price_Converged\"])\n",
|
||||
" .agg({\"Fiscal_Date\": min, \"Invoiced_quantity_in_Pieces\": sum, \"Net_Sales\": sum})\n",
|
||||
" .reset_index()\n",
|
||||
")\n",
|
||||
"# rename the aggregated columns\n",
|
||||
"df_aggr.rename(\n",
|
||||
" columns={\n",
|
||||
" \"Fiscal_Date\": \"First_price_date\",\n",
|
||||
" \"Invoiced_quantity_in_Pieces\": \"Total_ordered_pieces\",\n",
|
||||
" \"Net_Sales\": \"Total_net_sales\",\n",
|
||||
" },\n",
|
||||
" inplace=True,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# sort values chronologically\n",
|
||||
"df_aggr.sort_values(by=[\"Product_ID\", \"First_price_date\"], inplace=True)\n",
|
||||
"df_aggr.reset_index(drop=True, inplace=True)\n",
|
||||
"\n",
|
||||
"# add columns for previous values\n",
|
||||
"df_aggr[\"Previous_List\"] = df_aggr.groupby([\"Product_ID\"])[\n",
|
||||
" \"List_Price_Converged\"\n",
|
||||
"].shift()\n",
|
||||
"df_aggr[\"Previous_Total_ordered_pieces\"] = df_aggr.groupby([\"Product_ID\"])[\n",
|
||||
" \"Total_ordered_pieces\"\n",
|
||||
"].shift()\n",
|
||||
"\n",
|
||||
"# average price change across sku's\n",
|
||||
"df_aggr[\"price_change_perc\"] = (\n",
|
||||
" (df_aggr[\"List_Price_Converged\"] - df_aggr[\"Previous_List\"])\n",
|
||||
" / df_aggr[\"Previous_List\"].fillna(0)\n",
|
||||
" * 100\n",
|
||||
")\n",
|
||||
"df_aggr[\"order_change_perc\"] = (\n",
|
||||
" (df_aggr[\"Total_ordered_pieces\"] - df_aggr[\"Previous_Total_ordered_pieces\"])\n",
|
||||
" / df_aggr[\"Previous_Total_ordered_pieces\"].fillna(0)\n",
|
||||
" * 100\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# plot a scatterplot to visualize the changes\n",
|
||||
"sns.scatterplot(\n",
|
||||
" x=\"price_change_perc\",\n",
|
||||
" y=\"order_change_perc\",\n",
|
||||
" data=df_aggr,\n",
|
||||
" hue=\"Product_ID\",\n",
|
||||
" legend=False,\n",
|
||||
")\n",
|
||||
"plt.title(\"Percentage of change in price vs order\")\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "8259e916fe25"
|
||||
},
|
||||
"source": [
|
||||
"For most of the products, the percentage change in orders are high where the percentage changes in the prices are low. This suggests that too much change in the prices can affect the number of orders. \n",
|
||||
"\n",
|
||||
"**Note**: There seem to be some outliers in the data as percentage changes greater than 800 are found. In the current exercise, do not take any manual measures to deal with outliers as you will create a BigQuery ML timeseries model that already deals with outliers.\n",
|
||||
"\n",
|
||||
"## Preprocess the data for training\n",
|
||||
"<a name=\"section-8\"></a>\n",
|
||||
"\n",
|
||||
"Check which `Product_ID`'s have the maximum orders."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "f5cbc7709c6a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_orders = df.groupby([\"Product_ID\", \"Customer_Hierarchy\"], as_index=False)[\n",
|
||||
" \"Invoiced_quantity_in_Pieces\"\n",
|
||||
"].sum()\n",
|
||||
"df_orders.loc[\n",
|
||||
" df_orders.groupby(\"Customer_Hierarchy\")[\"Invoiced_quantity_in_Pieces\"].idxmax()\n",
|
||||
"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fd6d227e513e"
|
||||
},
|
||||
"source": [
|
||||
"From the above result, you can infer the following:\n",
|
||||
"\n",
|
||||
"- Under the **Food** category, **SKU 62** has the maximum orders.\n",
|
||||
"- Under the **Manufacturing** category, **SKU 17** has the maximum orders.\n",
|
||||
"- Under the **Paper** category, **SKU 107** has the maximum orders.\n",
|
||||
"- Under the **Publishing** category, **SKU 8** has the maximum orders.\n",
|
||||
"- Under the **Utilities** category, **SKU 140** has the maximum orders.\n",
|
||||
"\n",
|
||||
"Given that there are too many ids and only a few records for most of them, consider only the above `Product_ID`s for which there are a maximum number of orders. \n",
|
||||
"\n",
|
||||
"**Note**: The `Invoiced_quantity_in_Pieces` field seems to be a *float* type rather than an *int* type as it should be. This could be because the data itself might be averaged in the first place."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2dbc0d64d157"
|
||||
},
|
||||
"source": [
|
||||
"Check the various prices available for these `Product_ID`s."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "acc1dbd2d838"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_type_food = df[(df[\"Product_ID\"] == \"SKU 62\") & (df[\"Customer_Hierarchy\"] == \"Food\")]\n",
|
||||
"print(\"Food :\")\n",
|
||||
"print(df_type_food[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_manuf = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 17\") & (df[\"Customer_Hierarchy\"] == \"Manufacturing\")\n",
|
||||
"]\n",
|
||||
"print(\"Manufacturing :\")\n",
|
||||
"print(df_type_manuf[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_paper = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 107\") & (df[\"Customer_Hierarchy\"] == \"Paper\")\n",
|
||||
"]\n",
|
||||
"print(\"Paper :\")\n",
|
||||
"print(df_type_paper[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_pub = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 8\") & (df[\"Customer_Hierarchy\"] == \"Publishing\")\n",
|
||||
"]\n",
|
||||
"print(\"Publishing :\")\n",
|
||||
"print(df_type_pub[\"List_Price_Converged\"].value_counts())\n",
|
||||
"df_type_util = df[\n",
|
||||
" (df[\"Product_ID\"] == \"SKU 140\") & (df[\"Customer_Hierarchy\"] == \"Utilities\")\n",
|
||||
"]\n",
|
||||
"print(\"Utilities :\")\n",
|
||||
"print(df_type_util[\"List_Price_Converged\"].value_counts())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f023af578c0f"
|
||||
},
|
||||
"source": [
|
||||
"In the publishing category, `Product_ID` `SKU 8` and `SKU 17` are less than or equal to two different prices in the entire data and so you will exclude them and consider the rest for building the forecast model. The idea here is to train a forecast model on the timeseries data for products with different prices.\n",
|
||||
"\n",
|
||||
"Join the data for all the `Product_ID`s into one dataframe and remove duplicate records."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "a44771cc4c20"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"df_final = pd.concat([df_type_food, df_type_paper, df_type_util])\n",
|
||||
"df_final = (\n",
|
||||
" df_final[\n",
|
||||
" [\n",
|
||||
" \"Product_ID\",\n",
|
||||
" \"Fiscal_Date\",\n",
|
||||
" \"Customer_Hierarchy\",\n",
|
||||
" \"List_Price_Converged\",\n",
|
||||
" \"Invoiced_quantity_in_Pieces\",\n",
|
||||
" ]\n",
|
||||
" ]\n",
|
||||
" .drop_duplicates()\n",
|
||||
" .reset_index(drop=True)\n",
|
||||
")\n",
|
||||
"df_final.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "add5063df368"
|
||||
},
|
||||
"source": [
|
||||
"Save the data to a BigQuery table."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "fd82ba56571f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"bq_client = bigquery.Client(project=PROJECT_ID)\n",
|
||||
"\n",
|
||||
"job_config = bigquery.LoadJobConfig(\n",
|
||||
" # Specify a (partial) schema. All columns are always written to the\n",
|
||||
" # table. The schema is used to assist in data type definitions.\n",
|
||||
" schema=[\n",
|
||||
" bigquery.SchemaField(\"Product_ID\", bigquery.enums.SqlTypeNames.STRING),\n",
|
||||
" bigquery.SchemaField(\"Fiscal_Date\", bigquery.enums.SqlTypeNames.DATE),\n",
|
||||
" bigquery.SchemaField(\"List_Price_Converged\", bigquery.enums.SqlTypeNames.FLOAT),\n",
|
||||
" bigquery.SchemaField(\n",
|
||||
" \"Invoiced_quantity_in_Pieces\", bigquery.enums.SqlTypeNames.FLOAT\n",
|
||||
" ),\n",
|
||||
" ],\n",
|
||||
" # Optionally, set the write disposition. BigQuery appends loaded rows\n",
|
||||
" # to an existing table by default, but with WRITE_TRUNCATE write\n",
|
||||
" # disposition it replaces the table with the loaded data.\n",
|
||||
" write_disposition=\"WRITE_TRUNCATE\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# save the dataframe to a table in the created dataset\n",
|
||||
"job = bq_client.load_table_from_dataframe(\n",
|
||||
" df_final,\n",
|
||||
" \"{}.{}.{}\".format(PROJECT_ID, DATASET, TRAINING_DATA_TABLE),\n",
|
||||
" job_config=job_config,\n",
|
||||
") # Make an API request.\n",
|
||||
"job.result() # Wait for the job to complete."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fca77641b03b"
|
||||
},
|
||||
"source": [
|
||||
"# Train the model using BigQuery ML\n",
|
||||
"<a name=\"section-9\"></a>\n",
|
||||
"\n",
|
||||
"Train an [Arima-Plus](https://cloud.google.com/bigquery-ml/docs/reference/standard-sql/bigqueryml-syntax-create-time-series) model on the data using BigQuery ML."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cded27507891"
|
||||
},
|
||||
"source": [
|
||||
"#@bigquery\n",
|
||||
"create or replace model pricing_optimization.bqml_arima\n",
|
||||
"options\n",
|
||||
" (model_type = 'ARIMA_PLUS',\n",
|
||||
" time_series_timestamp_col = 'Fiscal_Date',\n",
|
||||
" time_series_data_col = 'Invoiced_quantity_in_Pieces',\n",
|
||||
" time_series_id_col = 'ID'\n",
|
||||
" ) as\n",
|
||||
"select\n",
|
||||
" Fiscal_Date,\n",
|
||||
" Concat(Product_ID,\"_\" ,Cast(List_Price_Converged as string)) as ID,\n",
|
||||
" Invoiced_quantity_in_Pieces\n",
|
||||
"from\n",
|
||||
" pricing_optimization.TRAINING_DATA\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "332fd11ff32b"
|
||||
},
|
||||
"source": [
|
||||
"## Generate forecasts from the model\n",
|
||||
"<a name=\"section-10\"></a>\n",
|
||||
"\n",
|
||||
"Predict the sales for the next 30 days for each id and save to a dataframe."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ef926cdbf28e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"client = Client()\n",
|
||||
"\n",
|
||||
"query = '''\n",
|
||||
"DECLARE HORIZON STRING DEFAULT \"30\"; #number of values to forecast\n",
|
||||
"DECLARE CONFIDENCE_LEVEL STRING DEFAULT \"0.90\"; ## required confidence level\n",
|
||||
"\n",
|
||||
"EXECUTE IMMEDIATE format(\"\"\"\n",
|
||||
" SELECT\n",
|
||||
" *\n",
|
||||
" FROM \n",
|
||||
" ML.FORECAST(MODEL pricing_optimization.bqml_arima, \n",
|
||||
" STRUCT(%s AS horizon, \n",
|
||||
" %s AS confidence_level)\n",
|
||||
" )\n",
|
||||
" \"\"\",HORIZON,CONFIDENCE_LEVEL)'''\n",
|
||||
"job = client.query(query)\n",
|
||||
"dfforecast = job.to_dataframe()\n",
|
||||
"dfforecast.head()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "608c7de72dae"
|
||||
},
|
||||
"source": [
|
||||
"## Interpret the results to choose the best price\n",
|
||||
"<a name=\"section-11\"></a>\n",
|
||||
"\n",
|
||||
"Calculate average forecast values for the forecast duration."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e1e193680400"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dfforecast_avg = (\n",
|
||||
" dfforecast[[\"ID\", \"forecast_value\"]].groupby(\"ID\", as_index=False).mean()\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5ce395d652a3"
|
||||
},
|
||||
"source": [
|
||||
"Extract the ID and Price fields from the ID field."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "452c56fa58ed"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dfforecast_avg[\"Product_ID\"] = dfforecast_avg[\"ID\"].apply(lambda x: x.split(\"_\")[0])\n",
|
||||
"dfforecast_avg[\"Price\"] = dfforecast_avg[\"ID\"].apply(lambda x: x.split(\"_\")[1])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3cee67f4028f"
|
||||
},
|
||||
"source": [
|
||||
"Plot the average forecasted sales vs. the price of the product."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "fb351c8f383d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"for i in dfforecast_avg[\"Product_ID\"].unique():\n",
|
||||
" dfforecast_avg[dfforecast_avg[\"Product_ID\"] == i].set_index(\"Price\").sort_values(\n",
|
||||
" \"forecast_value\"\n",
|
||||
" ).plot(kind=\"bar\")\n",
|
||||
" plt.title(\"Price vs. Average Sales for \" + i)\n",
|
||||
" plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "67ff3acc74a5"
|
||||
},
|
||||
"source": [
|
||||
"Based on the plots for price vs. the average forecasted orders, it can be said that to use the maximum orders, each of the considered `Product_ID`s can follow the below prices:\n",
|
||||
"\n",
|
||||
"- SKU 107's price range can be from 4.44 - 4.73 units\n",
|
||||
"- SKU 140's price can be 1.95 units\n",
|
||||
"- SKU 62's price can be 4.23 units\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Clean Up\n",
|
||||
"<a name=\"section-12\"></a>\n",
|
||||
"\n",
|
||||
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial. The following code deletes the entire dataset."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "d78908b8134d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Construct a BigQuery client object.\n",
|
||||
"client = bigquery.Client()\n",
|
||||
"\n",
|
||||
"# TODO(developer): Set model_id to the ID of the model to fetch.\n",
|
||||
"dataset_id = \"{PROJECT}.{DATASET}\".format(PROJECT=PROJECT_ID, DATASET=DATASET)\n",
|
||||
"\n",
|
||||
"# Use the delete_contents parameter to delete a dataset and its contents.\n",
|
||||
"# Use the not_found_ok parameter to not receive an error if the dataset has already been deleted.\n",
|
||||
"client.delete_dataset(\n",
|
||||
" dataset_id, delete_contents=True, not_found_ok=True\n",
|
||||
") # Make an API request.\n",
|
||||
"\n",
|
||||
"print(\"Deleted dataset '{}'.\".format(dataset_id))"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"name": "pricing-optimization.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -459,7 +459,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil cp gs://cloud-samples-data/ai-platform-unified/matching_engine/glove-100-angular.hdf5 ."
|
||||
"! gsutil cp gs://cloud-samples-data/vertex-ai/matching_engine/glove-100-angular.hdf5 ."
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -11,8 +11,8 @@ The purpose of this set of notebooks and markdown files is to demonstrate Google
|
||||
1. [Data Management](stage1)
|
||||
2. [Experimentation](stage2)
|
||||
3. [Formalization](stage3)
|
||||
4. Evaluation
|
||||
4. [Evaluation](stage4)
|
||||
5. Deployment
|
||||
6. Serving
|
||||
6. [Serving](stage6)
|
||||
7. Monitoring
|
||||
8. Continuous Training
|
||||
|
||||
@@ -30,10 +30,66 @@ The first stage in MLOps is the collection and preparation for the purpose of de
|
||||
|
||||
[Get Started with BQ datasets](get_started_bq_datasets.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Vertex AI `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.
|
||||
- Extract a copy of the dataset from `BigQuery` to a CSV file in Cloud Storage -- compatible for `AutoML` or custom training.
|
||||
- Select rows from a `BigQuery` dataset into a `pandas` dataframe -- compatible for custom training.
|
||||
- Select rows from a `BigQuery` dataset into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.
|
||||
- Select rows from extracted CSV files into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.
|
||||
- Create a `BigQuery` dataset from CSV files.
|
||||
- Extract data from `BigQuery` table into a `DMatrix` -- compatible for custom training `XGBoost` models.
|
||||
```
|
||||
|
||||
[Get Started with Vertex datasets](get_started_vertex_datasets.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a Vertex AI `Dataset` resource for:
|
||||
- image data
|
||||
- text data
|
||||
- video data
|
||||
- tabular data
|
||||
- forecasting data
|
||||
|
||||
|
||||
- Search `Dataset` resources using a filter.
|
||||
- Read a sample of a `BigQuery` dataset into a dataframe.
|
||||
- Generate statistics and data schema using TensorFlow Data Validation from the samples in the dataframe.
|
||||
- Detect anomalies in new data using TensorFlow Data Validation.
|
||||
- Generate a TFRecord feature specification using TensorFlow Transform from the data schema.
|
||||
- Export a dataset and convert to TFRecords.
|
||||
```
|
||||
|
||||
[Get Started with Dataflow](get_started_dataflow.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Offline preprocessing of data:
|
||||
- Serially - w/o dataflow
|
||||
- Parallel - with dataflow
|
||||
- Upstream preprocessing of data:
|
||||
- tabular data
|
||||
- image data
|
||||
```
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 1: Data Management](mlops_data_management.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Explore and visualize the data.
|
||||
- Create a Vertex AI `Dataset` resource from `BigQuery` table -- for AutoML training.
|
||||
- Extract a copy of the dataset to a CSV file in Cloud Storage.
|
||||
- Create a Vertex AI `Dataset` resource from CSV files -- alternative for AutoML training.
|
||||
- Read a sample of the `BigQuery` dataset into a dataframe.
|
||||
- Generate statistics and data schema using TensorFlow Data Validation from the samples in the dataframe.
|
||||
- Generate a TFRecord feature specification using TensorFlow Data Validation from the data schema.
|
||||
- Preprocess a portion of the BigQuery data using `Dataflow` -- for custom training.
|
||||
```
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2020 Google LLC\n",
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_bq_datasets.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -82,12 +82,12 @@
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex Datasets`\n",
|
||||
"- `Vertex AI Datasets`\n",
|
||||
"- `BigQuery Datasets`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Create a Vertex `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.\n",
|
||||
"- Create a Vertex AI `Dataset` resource from `BigQuery` table -- compatible for `AutoML` training.\n",
|
||||
"- Extract a copy of the dataset from `BigQuery` to a CSV file in Cloud Storage -- compatible for `AutoML` or custom training.\n",
|
||||
"- Select rows from a `BigQuery` dataset into a `pandas` dataframe -- compatible for custom training.\n",
|
||||
"- Select rows from a `BigQuery` dataset into a `tf.data.Dataset` -- compatible for custom training `TensorFlow` models.\n",
|
||||
@@ -107,7 +107,7 @@
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with structured (tabular) data in BigQuery:\n",
|
||||
"\n",
|
||||
"- For AutoML training:\n",
|
||||
" - Create a managed dataset with Vertex `TabularDataset`.\n",
|
||||
" - Create a managed dataset with Vertex AI `TabularDataset`.\n",
|
||||
" - Use the BigQuery table as the input to the dataset.\n",
|
||||
" - Specify columns and columns transformations when running the AutoML training pipeline job.\n",
|
||||
"\n",
|
||||
@@ -164,11 +164,13 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG"
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -285,7 +287,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -501,9 +503,9 @@
|
||||
"id": "init_aip:mbsdk,region"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex SDK for Python\n",
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1188,6 +1190,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_dataflow.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_dataflow.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -157,11 +157,13 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG"
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -258,7 +260,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -474,7 +476,7 @@
|
||||
"id": "import_numpy"
|
||||
},
|
||||
"source": [
|
||||
"#### Import pandas\n",
|
||||
"#### Import numpy\n",
|
||||
"\n",
|
||||
"Import the numpy package into your Python environment."
|
||||
]
|
||||
@@ -540,9 +542,9 @@
|
||||
"id": "init_aip:mbsdk,region"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex SDK for Python\n",
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -674,6 +676,31 @@
|
||||
"dataframe[\"station_number\"] = pd.to_numeric(dataframe[\"station_number\"])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"source": [
|
||||
"### Create BQ dataset resource\n",
|
||||
"\n",
|
||||
"First, you create an empty dataset resource in your project."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "bqml_create_dataset"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BQ_MY_DATASET = 'samples'\n",
|
||||
"BQ_MY_TABLE = 'gsod'\n",
|
||||
"! bq --location=US mk -d \\\n",
|
||||
"$PROJECT_ID:$BQ_MY_DATASET"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -1201,6 +1228,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/get_started_vertex_datasets.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -67,16 +67,16 @@
|
||||
"source": [
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this tutorial, you learn how to use `Vertex Dataset` for training with `Vertex AI`.\n",
|
||||
"In this tutorial, you learn how to use `Vertex AI Dataset` for training with `Vertex AI`.\n",
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex Datasets`\n",
|
||||
"- `Vertex AI Datasets`\n",
|
||||
"- `BigQuery Datasets`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Create a Vertex `Dataset` resource for:\n",
|
||||
"- Create a Vertex AI `Dataset` resource for:\n",
|
||||
" - image data\n",
|
||||
" - text data\n",
|
||||
" - video data\n",
|
||||
@@ -100,7 +100,7 @@
|
||||
"source": [
|
||||
"### Recommendations\n",
|
||||
"\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with Vertex Datasets:\n",
|
||||
"When doing E2E MLOps on Google Cloud, the following best practices with Vertex AI Datasets:\n",
|
||||
"\n",
|
||||
"- Use CSV index file format for image data\n",
|
||||
"- Use CSV index file format for text data:\n",
|
||||
@@ -116,7 +116,7 @@
|
||||
"\n",
|
||||
"- Use `filter` and `order_by` parameters in the `list()` methods to find the latest versions of datasets.\n",
|
||||
"\n",
|
||||
"- When custom training with a `Vertex Dataset`:\n",
|
||||
"- When custom training with a `Vertex AI Dataset`:\n",
|
||||
" - tabular data :\n",
|
||||
" - Use the CSV index file or BigQuery table reference.\n",
|
||||
" - Create a tf.data.Dataset generator from the CSV index file/BigQuery table.\n",
|
||||
@@ -159,11 +159,13 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG"
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -260,7 +262,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -498,9 +500,9 @@
|
||||
"id": "init_aip:mbsdk,region"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex SDK for Python\n",
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -542,9 +544,9 @@
|
||||
"id": "dataset_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Vertex Datasets\n",
|
||||
"## Vertex AI Datasets\n",
|
||||
"\n",
|
||||
"Vertex `Datasets` are the means for managing your datasets within Vertex AI services. Vertex Datasets are also referred to as `Dataset` resources. There are four types of `Dataset` resources, specific to the data type:\n",
|
||||
"Vertex AI `Datasets` are the means for managing your datasets within Vertex AI services. Vertex AI Datasets are also referred to as `Dataset` resources. There are four types of `Dataset` resources, specific to the data type:\n",
|
||||
"\n",
|
||||
"- `ImageDataset`: image data\n",
|
||||
"- `TabularDataset`: tabular (structured) data\n",
|
||||
@@ -552,7 +554,7 @@
|
||||
"- `VideoDataset`: video data\n",
|
||||
"- `TimeSeriesDataset`: forecasting data\n",
|
||||
"\n",
|
||||
"A Vertex `Dataset` provides the following capabilities:\n",
|
||||
"A Vertex AI `Dataset` provides the following capabilities:\n",
|
||||
"\n",
|
||||
"- A unique internal identifier for automatic (programatic) processes.\n",
|
||||
"- A user specificed (display name) identifier for interactive processes.\n",
|
||||
@@ -809,7 +811,7 @@
|
||||
"id": "dataset_intro:methods"
|
||||
},
|
||||
"source": [
|
||||
"## Vertex Dataset properties and methods\n",
|
||||
"## Vertex AI Dataset properties and methods\n",
|
||||
"\n",
|
||||
"The following are the `Dataset` methods:\n",
|
||||
"\n",
|
||||
@@ -898,7 +900,7 @@
|
||||
"source": [
|
||||
"## TensorFlow Data Validation\n",
|
||||
"\n",
|
||||
"The TensorFlow Data Validation (TFDV) package is used in conjunction with Vertex and BigQuery datasets for:\n",
|
||||
"The TensorFlow Data Validation (TFDV) package is used in conjunction with Vertex AI and BigQuery datasets for:\n",
|
||||
"\n",
|
||||
"- Generating dataset statistics.\n",
|
||||
"- Generating data schema for data validation.\n",
|
||||
@@ -1568,6 +1570,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/mlops_data_management.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/mlops_data_management.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage1/mlops_data_management.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage1/mlops_data_management.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -140,6 +140,7 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -242,7 +243,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -33,18 +33,180 @@ The second stage in MLOps is experimenting in developing one or more baseline mo
|
||||
|
||||
[Get Started with Vertex Experiments and Vertex ML Metadata](get_started_vertex_experiments.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Use Python logging to log training configuration/results locally.
|
||||
- Use Google Cloud Logging to log training configuration/results in cloud storage.
|
||||
- Create a Vertex AI `Experiment` resource.
|
||||
- Instantiate an experiment run.
|
||||
- Log parameters for the run.
|
||||
- Log metrics for the run.
|
||||
- Display the logged experiment run.
|
||||
```
|
||||
|
||||
[Get Started with Vertex TensorBoard](get_started_vertex_tensorboard.ipynb)
|
||||
|
||||
[Get Started with Custom Training Packages](get_started_vertex_training.ipynb)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a TensorBoard callback when training a model.
|
||||
- Using Tensorboard with locally trained model.
|
||||
- Using Vertex AI TensorBoard with Vertex AI Training.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Tensorflow)](get_started_vertex_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a single Python script.
|
||||
- Training using a Python package.
|
||||
- Training using a custom training image.
|
||||
- Laying out a training package.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Scikit-Learn)](get_started_vertex_training_sklearn.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (XGBoost)](get_started_vertex_training_xgboost.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (Pytorch)](get_started_vertex_training_pytorch.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Single node training using a Python package.
|
||||
- Report accuracy when hyperparameter tuning.
|
||||
- Save the model artifacts to Cloud Storage using GCSFuse.
|
||||
- Create a `Vertex AI Model` resource.
|
||||
```
|
||||
|
||||
[Get Started with Custom Training Packages (R)](get_started_vertex_training_r.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Locally train an R model in a notebook using %%R magic commands
|
||||
- Create a deployment image with trained R model and serving functions.
|
||||
- Test the deployment image locally.
|
||||
- Create a `Vertex AI Model` resource for the deployment image with embedded R model.
|
||||
- Deploy the deployment image with embedded R model to a `Vertex AI Endpoint` resource.
|
||||
- Test the deployment image with embedded R model.
|
||||
- Create a R-to-Python training package.
|
||||
- Create a training image for training the model.
|
||||
- Train a R model using `Vertex AI Trainingh` service with the R-to-Python training package.
|
||||
```
|
||||
|
||||
[Get Started with Distributed Training](get_started_vertex_distributed_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- `MirroredStrategy`: Train on a single VM with multiple GPUs.
|
||||
- `MultiWorkerMirroredStrategy`: Train on multiple VMs with automatic setup of replicas.
|
||||
- `MultiWorkerMirroredStrategy`: Train on multiple VMs with fine grain control of replicas.
|
||||
- `ReductionServer`: Train on multiple VMS and sync updates across VMS with `Vertex AI Reduction Server`.
|
||||
- `TPUTraining`: Train with multiple Cloud TPUs.
|
||||
```
|
||||
|
||||
[Get Started with Vizier Hyperparameter Tuning](get_started_vertex_vizier.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Hyperparameter tuning with Random algorithm.
|
||||
- Hyperparameter tuning with Vizier (Bayesian) algorithm.
|
||||
```
|
||||
|
||||
[Get Started with AutoML Training](get_started_automl_training.ipynb)
|
||||
|
||||
[Get Started with BQML Training](get_started_bqml_training.ipyn)
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Train an image model.
|
||||
- Export the image model as an edge model.
|
||||
- Train a tabular model.
|
||||
- Export the tabular model as a cloud model.
|
||||
- Train a text model.
|
||||
```
|
||||
|
||||
[Get Started with BQML Training](get_started_bqml_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Create a local BQ table in your project.
|
||||
- Train a BQML model.
|
||||
- Evaluate the BQML model.
|
||||
- Export the BQML model as a cloud model.
|
||||
- Upload the exported model as a Vertex AI Model resource.
|
||||
- Hyperparameter tune a BQML model with Vertex AI Vizier.
|
||||
```
|
||||
|
||||
[Get Started with Vertex Feature Store](get_started_vertex_feature_store.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a Vertex AI `Featurestore` resource.
|
||||
- Creating `EntityType` resources for the `Featurestore` resource.
|
||||
- Creating `Feature` resources for each `EntityType` resource.
|
||||
- Import feature values (entity data items) into `Featurestore` resource from Cloud Storage.
|
||||
- Import feature values (entity data items) into `Featurestore` resource from pandas DataFrame.
|
||||
- Perform online serving from a `Featurestore` resource.
|
||||
- Perform batch serving from a `Featurestore` resource.
|
||||
```
|
||||
|
||||
[Get Started with Google CMEK Training](get_started_with_cmek_training.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a customer managed encryption key.
|
||||
- Creating an image dataset with CMEK encryption.
|
||||
- Train an AutoML model with CMEK encryption.
|
||||
```
|
||||
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 2: Experimentation](mlops_experimentation.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Review the `Dataset` resource created during stage 1.
|
||||
- Train an AutoML tabular binary classifier model in the background.
|
||||
- Build the experimental model architecture.
|
||||
- Construct a custom training package for the `Dataset` resource.
|
||||
- Test the custom training package locally.
|
||||
- Test the custom training package in the cloud with Vertex AI Training.
|
||||
- Hyperparameter tune the model training with Vertex AI Vizier.
|
||||
- Train the custom model with Vertex AI Training.
|
||||
- Add a serving function for online/batch prediction to the custom model.
|
||||
- Test the custom model with the serving function.
|
||||
- Evaluate the custom model using Vertex AI Batch Prediction
|
||||
- Wait for the AutoML training job to complete.
|
||||
- Evaluate the AutoML model using Vertex AI Batch Prediction with the same evaluation slices as the custom model.
|
||||
- Set the evaluation results of the AutoML model as the baseline.
|
||||
- If the evaluation of the custom model is below baseline, continue to experiment with the custom model.
|
||||
- If the evaluation of the custom model is above baseline, save the model as the first best model.
|
||||
```
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_automl_training.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_automl_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -156,6 +156,7 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -258,7 +259,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_distributed_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -87,10 +87,11 @@
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- `MirroredStrategy`: Train on single VM with multiple GPUs.\n",
|
||||
"- `MirroredStrategy`: Train on a single VM with multiple GPUs.\n",
|
||||
"- `MultiWorkerMirroredStrategy`: Train on multiple VMs with automatic setup of replicas.\n",
|
||||
"- `MultiWorkerMirroredStrategy`: Train on multiple VMs with fine grain control of replicas.\n",
|
||||
"- `ReductionServer`: Train on multiple VMS and sync updates across VMS with Vertex AI Reduction Server"
|
||||
"- `ReductionServer`: Train on multiple VMS and sync updates across VMS with `Vertex AI Reduction Server`.\n",
|
||||
"- `TPUTraining`: Train with multiple Cloud TPUs."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -148,12 +149,15 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -250,7 +254,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -428,9 +432,9 @@
|
||||
"\n",
|
||||
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
|
||||
"\n",
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region\n",
|
||||
"Learn more about [hardware accelerator support for your region](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators).\n",
|
||||
"\n",
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3. This is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1543,6 +1547,233 @@
|
||||
"job.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "tpu_intro"
|
||||
},
|
||||
"source": [
|
||||
"## Cloud TPU Training\n",
|
||||
"\n",
|
||||
"To further speed up trainig, your organization can utilize Google's Cloud Tensor Processing Units (TPU) pods.\n",
|
||||
"\n",
|
||||
"Cloud TPU is the custom-designed machine learning ASIC that powers Google products like Translate, Photos, Search, Assistant, and Gmail. Cloud TPU is designed to run cutting-edge machine learning models with AI services on Google Cloud. And its custom high-speed network offers over 100 petaflops of performance in a single pod.\n",
|
||||
"\n",
|
||||
"Learn more about [Cloud TPU](https://cloud.google.com/tpu)\n",
|
||||
"\n",
|
||||
"*Note*: TPU VM Training is currently an opt-in feature. Your GCP project must first be added to the feature allowlist. Please email your project information(project id/number) to vertex-ai-tpu-vm-training-support@google.com for the allowlist. You will receive an email as soon as your project is ready."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "docker_write:tpu"
|
||||
},
|
||||
"source": [
|
||||
"### Write Docker file for TPU training\n",
|
||||
"\n",
|
||||
"Currently, there is no pre-built Vertex AI Docker image for training with TPUs. No problems, you can make your own, as follows:\n",
|
||||
"\n",
|
||||
"1. Create a vanilla Python 3 image (e.g., `python3:8`).\n",
|
||||
"2. Get and install the TPU library (`libtpu.so`).\n",
|
||||
"3. Copy in your training package"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "docker_write:tpu"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile custom/Dockerfile\n",
|
||||
"FROM python:3.8\n",
|
||||
"\n",
|
||||
"WORKDIR /root\n",
|
||||
"\n",
|
||||
"# Copies the trainer code to the docker image.\n",
|
||||
"COPY trainer /trainer\n",
|
||||
"\n",
|
||||
"RUN pip3 install tensorflow-datasets\n",
|
||||
"\n",
|
||||
"# Install TPU Tensorflow and dependencies.\n",
|
||||
"# libtpu.so must be under the '/lib' directory.\n",
|
||||
"RUN wget https://storage.googleapis.com/cloud-tpu-tpuvm-artifacts/libtpu/20210525/libtpu.so -O /lib/libtpu.so\n",
|
||||
"RUN chmod 777 /lib/libtpu.so\n",
|
||||
"\n",
|
||||
"RUN wget https://storage.googleapis.com/cloud-tpu-tpuvm-artifacts/tensorflow/20210525/tf_nightly-2.6.0-cp38-cp38-linux_x86_64.whl\n",
|
||||
"RUN pip3 install tf_nightly-2.6.0-cp38-cp38-linux_x86_64.whl\n",
|
||||
"RUN rm tf_nightly-2.6.0-cp38-cp38-linux_x86_64.whl\n",
|
||||
"# Sets up the entry point to invoke the trainer.\n",
|
||||
"ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "docker_push:tpu"
|
||||
},
|
||||
"source": [
|
||||
"### Build and push the Docker image to the Artifact Registry"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "docker_push:tpu"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"TRAIN_IMAGE = f\"gcr.io/\" + PROJECT_ID + \"/tpu-train:latest\"\n",
|
||||
"\n",
|
||||
"os.chdir(\"custom\")\n",
|
||||
"! docker build --quiet --tag={TRAIN_IMAGE} .\n",
|
||||
"! docker push {TRAIN_IMAGE}\n",
|
||||
"os.chdir(\"..\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "worker_pool_tpu"
|
||||
},
|
||||
"source": [
|
||||
"### TPU worker specification pool\n",
|
||||
"\n",
|
||||
"Next, you create the worker specification pool. For TPUs, you do:\n",
|
||||
"\n",
|
||||
"- Create only one worker pool (Primary).\n",
|
||||
"- Set the machine type to `cloud-tpu`.\n",
|
||||
"- Set the accelerator type to a `TPU`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "worker_pool_tpu"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Use TPU Accelerators. Temporarily using numeric codes, until types are added to the SDK\n",
|
||||
"# 6 = TPU_V2\n",
|
||||
"# 7 = TPU_V3\n",
|
||||
"TRAIN_TPU, TRAIN_NTPU = (7, 8)\n",
|
||||
"TRAIN_COMPUTE = \"cloud-tpu\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"if not TRAIN_NTPU or TRAIN_NTPU < 2:\n",
|
||||
" TRAIN_STRATEGY = \"single\"\n",
|
||||
"else:\n",
|
||||
" TRAIN_STRATEGY = \"tpu\"\n",
|
||||
"print(TRAIN_STRATEGY)\n",
|
||||
"\n",
|
||||
"EPOCHS = 20\n",
|
||||
"STEPS = 10000\n",
|
||||
"\n",
|
||||
"TRAINER_ARGS = [\n",
|
||||
" \"--epochs=\" + str(EPOCHS),\n",
|
||||
" \"--steps=\" + str(STEPS),\n",
|
||||
" \"--distribute=\" + TRAIN_STRATEGY,\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"WORKER_POOL_SPECS = [\n",
|
||||
" {\n",
|
||||
" \"container_spec\": {\n",
|
||||
" \"args\": TRAINER_ARGS,\n",
|
||||
" \"image_uri\": TRAIN_IMAGE,\n",
|
||||
" },\n",
|
||||
" \"replica_count\": 1,\n",
|
||||
" \"machine_spec\": {\n",
|
||||
" \"machine_type\": TRAIN_COMPUTE,\n",
|
||||
" \"accelerator_type\": TRAIN_TPU,\n",
|
||||
" \"accelerator_count\": TRAIN_NTPU,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"]\n",
|
||||
"\n",
|
||||
"print(WORKER_POOL_SPECS[0])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "custom_job:worker_pool"
|
||||
},
|
||||
"source": [
|
||||
"### Create CustomJob with worker pool specifications\n",
|
||||
"\n",
|
||||
"Next, you create a `CustomJob` for the multi-worker distributed training job:\n",
|
||||
"\n",
|
||||
"-`display_name`: The display name for the custom job.\n",
|
||||
"\n",
|
||||
"-`worker_pool_specs`: The detailed specifications for each worker pool."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "custom_job:worker_pool"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"DISPLAY_NAME = \"boston_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"job = aip.CustomJob(display_name=DISPLAY_NAME, worker_pool_specs=WORKER_POOL_SPECS)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "run_custom_job:multiworker"
|
||||
},
|
||||
"source": [
|
||||
"### Run the CustomJob\n",
|
||||
"\n",
|
||||
"Next, you run the custom job."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_job:multiworker"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"try:\n",
|
||||
" job.run(sync=True)\n",
|
||||
"except Exception as e:\n",
|
||||
" # may fail in multi-worker to find startup script\n",
|
||||
" print(e)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a custom training job\n",
|
||||
"\n",
|
||||
"After a training job is completed, you can delete the training job with the method `delete()`. Prior to completion, a training job can be canceled with the method `cancel()`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_job"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_experiments.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -134,6 +134,7 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -236,7 +237,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -716,11 +717,36 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPERIMENT_NAME = \"example\"\n",
|
||||
"\n",
|
||||
"experiment_df = aip.get_experiment_df()\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == \"example\"]\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == EXPERIMENT_NAME]\n",
|
||||
"experiment_df.T"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_experiment"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the experiment\n",
|
||||
"\n",
|
||||
"Next, delete the experiment. You will need to get the context via the metadata to delete it."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_experiment"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"c = aiplatform.metadata._Context(EXPERIMENT_NAME)\n",
|
||||
"c.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -773,6 +799,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_tensorboard.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -130,6 +130,7 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -232,7 +233,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1107,6 +1108,28 @@
|
||||
"Alternatively, you can navigate to the Experiments tab and view the list of all experiments. Your experiment will have the same name as the training job."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_tensorboard"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the TensorBoard instance\n",
|
||||
"\n",
|
||||
"Next, delete the TensorBoard instance."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_tensorboard"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"tensorboard.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -1159,6 +1182,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -145,6 +145,7 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -247,7 +248,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -425,9 +426,9 @@
|
||||
"\n",
|
||||
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
|
||||
"\n",
|
||||
"Learn more [here](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators) hardware accelerator support for your region\n",
|
||||
"Learn more about [hardware accelerator support for your region](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators).\n",
|
||||
"\n",
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3. This is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -622,7 +623,7 @@
|
||||
"In summary:\n",
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Write \"hello world\" to the file."
|
||||
]
|
||||
},
|
||||
@@ -847,7 +848,7 @@
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Get the number of epochs to run from the command line (`--model_dir`).\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Repeat appending \"hello world\" to the file, one per epoch."
|
||||
]
|
||||
},
|
||||
@@ -1044,7 +1045,7 @@
|
||||
"\n",
|
||||
"- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"- Get the number of epochs to run from the command line (`--model_dir`).\n",
|
||||
"- Open a file \"test.txt\" in the directory where to sace the model artifacts.\n",
|
||||
"- Open a file \"test.txt\" in the directory where to save the model artifacts.\n",
|
||||
"- Repeat appending \"hello world\" to the file, one per epoch."
|
||||
]
|
||||
},
|
||||
@@ -1798,6 +1799,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,16 +33,17 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_vizier.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_vizier.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/get_started_vertex_vizier.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_vertex_vizier.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
" Open in Vertex AI Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" </td> \n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
@@ -139,6 +140,25 @@
|
||||
"Install *one time* the packages for executing the MLOps notebooks."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "1fd00fa70a2a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"# The Google Cloud Notebook product has specific requirements\n",
|
||||
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
|
||||
"\n",
|
||||
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
|
||||
"USER_FLAG = \"\"\n",
|
||||
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
|
||||
" USER_FLAG = \"--user\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -147,19 +167,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ONCE_ONLY = False\n",
|
||||
"if ONCE_ONLY:\n",
|
||||
" ! pip3 install -U tensorflow==2.5 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-data-validation==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
"! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" "
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -256,7 +265,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -267,7 +276,9 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
"REGION = \"[your-region]\" # @param {type:\"string\"}\n",
|
||||
"if REGION == \"[your-region]\":\n",
|
||||
" REGION = \"us-central1\""
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -317,7 +328,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -328,8 +339,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -349,7 +360,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
"! gsutil mb -l $REGION $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -369,7 +380,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
"! gsutil ls -al $BUCKET_URI"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -414,7 +425,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
|
||||
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -807,7 +818,7 @@
|
||||
"! rm -f custom.tar custom.tar.gz\n",
|
||||
"! tar cvf custom.tar custom\n",
|
||||
"! gzip custom.tar\n",
|
||||
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz"
|
||||
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_boston.tar.gz"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -915,7 +926,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"JOB_NAME = \"custom_job_\" + TIMESTAMP\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n",
|
||||
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, JOB_NAME)\n",
|
||||
"\n",
|
||||
"if not TRAIN_NGPU or TRAIN_NGPU < 2:\n",
|
||||
" TRAIN_STRATEGY = \"single\"\n",
|
||||
@@ -947,7 +958,7 @@
|
||||
" \"disk_spec\": disk_spec,\n",
|
||||
" \"python_package_spec\": {\n",
|
||||
" \"executor_image_uri\": TRAIN_IMAGE,\n",
|
||||
" \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n",
|
||||
" \"package_uris\": [BUCKET_URI + \"/trainer_boston.tar.gz\"],\n",
|
||||
" \"python_module\": \"trainer.task\",\n",
|
||||
" \"args\": CMDARGS,\n",
|
||||
" },\n",
|
||||
@@ -1576,14 +1587,6 @@
|
||||
"\n",
|
||||
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
|
||||
"\n",
|
||||
"- Dataset\n",
|
||||
"- Pipeline\n",
|
||||
"- Model\n",
|
||||
"- Endpoint\n",
|
||||
"- AutoML Training Job\n",
|
||||
"- Batch Job\n",
|
||||
"- Custom Job\n",
|
||||
"- Hyperparameter Tuning Job\n",
|
||||
"- Cloud Storage Bucket"
|
||||
]
|
||||
},
|
||||
@@ -1595,60 +1598,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"delete_all = True\n",
|
||||
"\n",
|
||||
"if delete_all:\n",
|
||||
" # Delete the dataset using the Vertex dataset object\n",
|
||||
" try:\n",
|
||||
" if \"dataset\" in globals():\n",
|
||||
" dataset.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the model using the Vertex model object\n",
|
||||
" try:\n",
|
||||
" if \"model\" in globals():\n",
|
||||
" model.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the AutoML or Pipeline training job\n",
|
||||
" try:\n",
|
||||
" if \"dag\" in globals():\n",
|
||||
" dag.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the custom training job\n",
|
||||
" try:\n",
|
||||
" if \"job\" in globals():\n",
|
||||
" job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the batch prediction job using the Vertex batch prediction object\n",
|
||||
" try:\n",
|
||||
" if \"batch_predict_job\" in globals():\n",
|
||||
" batch_predict_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
|
||||
" try:\n",
|
||||
" if \"hpt_job\" in globals():\n",
|
||||
" hpt_job.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
"\n",
|
||||
" if \"BUCKET_NAME\" in globals():\n",
|
||||
" ! gsutil rm -r $BUCKET_NAME"
|
||||
"if os.getenv(\"IS_TESTING\"):\n",
|
||||
" ! gsutil rm -r $BUCKET_URI"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -0,0 +1,874 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "VBOfRw7ifk8w"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
"# You may obtain a copy of the License at\n",
|
||||
"#\n",
|
||||
"# https://www.apache.org/licenses/LICENSE-2.0\n",
|
||||
"#\n",
|
||||
"# Unless required by applicable law or agreed to in writing, software\n",
|
||||
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
|
||||
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
|
||||
"# See the License for the specific language governing permissions and\n",
|
||||
"# limitations under the License."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "title:generic,gcp"
|
||||
},
|
||||
"source": [
|
||||
"# E2E ML on GCP: MLOps stage 2 : AutoML Image Classfication Training with Customer Managed Encryption Keys (CMEK)\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_with_cmek_training.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_with_cmek_training.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>\n",
|
||||
"<br/><br/><br/>"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "overview:mlops"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"This tutorial demonstrates how to use Vertex AI for E2E MLOps on Google Cloud in production. This tutorial covers stage 2 : experimentation: get started with AutoML training with a customer managed encyrption key CMEK."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "dataset:flowers,icn"
|
||||
},
|
||||
"source": [
|
||||
"### Dataset\n",
|
||||
"\n",
|
||||
"The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public #(GCS) bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip.\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "objective:mlops,stage3,get_started_automl_pipeline_components"
|
||||
},
|
||||
"source": [
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"In this tutorial, you learn how to use a customer managed encryption key (CMEK) for `Vertex AI AutoML` training.\n",
|
||||
"\n",
|
||||
"This tutorial uses the following Google Cloud ML services:\n",
|
||||
"\n",
|
||||
"- `Vertex AI AutoML`\n",
|
||||
"- Customer managed encryption key.\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Creating a customer managed encryption key.\n",
|
||||
"- Creating an image dataset with CMEK encryption.\n",
|
||||
"- Train an AutoML model with CMEK encryption."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "install_mlops"
|
||||
},
|
||||
"source": [
|
||||
"## Installations\n",
|
||||
"\n",
|
||||
"Install the Vertex AI SDK and the KMS package for CMEK encryption."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "sBfZtR4X1Dr_"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"USER_FLAG = \"--user\"\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG\n",
|
||||
"! pip3 install --upgrade google-cloud-kms $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
},
|
||||
"source": [
|
||||
"### Restart the kernel\n",
|
||||
"\n",
|
||||
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "restart"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"\n",
|
||||
"if not os.getenv(\"IS_TESTING\"):\n",
|
||||
" # Automatically restart kernel after installs\n",
|
||||
" import IPython\n",
|
||||
"\n",
|
||||
" app = IPython.Application.instance()\n",
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "project_id"
|
||||
},
|
||||
"source": [
|
||||
"#### Set your project ID\n",
|
||||
"\n",
|
||||
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
|
||||
" PROJECT_ID = shell_output[0]\n",
|
||||
" print(\"Project ID:\", PROJECT_ID)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "set_gcloud_project_id"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud config set project $PROJECT_ID"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
},
|
||||
"source": [
|
||||
"#### Region\n",
|
||||
"\n",
|
||||
"You can also change the `REGION` variable, which is used for operations\n",
|
||||
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
|
||||
"\n",
|
||||
"- Americas: `us-central1`\n",
|
||||
"- Europe: `europe-west4`\n",
|
||||
"- Asia Pacific: `asia-east1`\n",
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "region"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"REGION = \"us-central1\" # @param {type: \"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
},
|
||||
"source": [
|
||||
"#### Timestamp\n",
|
||||
"\n",
|
||||
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "timestamp"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bucket:mbsdk"
|
||||
},
|
||||
"source": [
|
||||
"### Create a Cloud Storage bucket\n",
|
||||
"\n",
|
||||
"**The following steps are required, regardless of your notebook environment.**\n",
|
||||
"\n",
|
||||
"When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
|
||||
"\n",
|
||||
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "autoset_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
|
||||
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
},
|
||||
"source": [
|
||||
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil mb -l $REGION $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
},
|
||||
"source": [
|
||||
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "validate_bucket"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gsutil ls -al $BUCKET_NAME"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "setup_vars"
|
||||
},
|
||||
"source": [
|
||||
"### Set up variables\n",
|
||||
"\n",
|
||||
"Next, set up some variables used throughout the tutorial.\n",
|
||||
"### Import libraries and define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_aip:mbsdk"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import google.cloud.aiplatform as aip\n",
|
||||
"from google.cloud import kms"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_NAME)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "mRk9eoTm6Pyi"
|
||||
},
|
||||
"source": [
|
||||
"## Setting up Customer Managed Encryption Keys\n",
|
||||
"\n",
|
||||
"By default, Google Cloud automatically encrypts data when it is stored in Cloud Storage using encryption keys managed by Google. If you have specific compliance or regulatory requirements related to the keys that protect your data, you can use customer-managed encryption keys (CMEK) for your training jobs.\n",
|
||||
"\n",
|
||||
"### Enable KMS API\n",
|
||||
"\n",
|
||||
"First, you enble the [Cloud Key Management Service (KMS)](https://console.cloud.google.com/flows/enableapi?apiid=cloudkms.googleapis.com)\n",
|
||||
"\n",
|
||||
"Learn more about [Customer managed encryption keys (CMEK)](https://cloud.google.com/vertex-ai/docs/general/cmek)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "RD_Pvrg584X3"
|
||||
},
|
||||
"source": [
|
||||
"### Create a key ring\n",
|
||||
"\n",
|
||||
"After you have enabled the KMS API, you create a key ring and a key. Use the helper function `create_key_ring()` to create a key ring, with the following parameters:\n",
|
||||
"\n",
|
||||
"- `project_id`: Your project ID.\n",
|
||||
"- `location`: Your region.\n",
|
||||
"- `key_ring_id`: The unique identifier for your key ring.\n",
|
||||
"\n",
|
||||
"The helper function calls the KMS client method `create_key_ring()` to create your key ring.\n",
|
||||
"\n",
|
||||
"Learn more about [KMS: Create a key ring](https://cloud.google.com/kms/docs/samples/kms-create-key-ring)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "dxRZzbvQnZC7"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"KEY_RING_ID = \"your_cmek_key_ring_id\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_key_ring(project_id, location, key_ring_id):\n",
|
||||
" \"\"\"\n",
|
||||
" Creates a new key ring in Cloud KMS\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" project_id (string): Google Cloud project ID (e.g. 'my-project').\n",
|
||||
" location (string): Cloud KMS location (e.g. 'us-east1').\n",
|
||||
" id (string): ID of the key ring to create (e.g. 'my-key-ring').\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" KeyRing: Cloud KMS key ring.\n",
|
||||
"\n",
|
||||
" \"\"\"\n",
|
||||
"\n",
|
||||
" # Create the client.\n",
|
||||
" client = kms.KeyManagementServiceClient()\n",
|
||||
"\n",
|
||||
" # Build the parent location name.\n",
|
||||
" location_name = f\"projects/{project_id}/locations/{location}\"\n",
|
||||
"\n",
|
||||
" # Build the key ring.\n",
|
||||
" key_ring = {}\n",
|
||||
"\n",
|
||||
" # Call the API.\n",
|
||||
" created_key_ring = client.create_key_ring(\n",
|
||||
" request={\n",
|
||||
" \"parent\": location_name,\n",
|
||||
" \"key_ring_id\": key_ring_id,\n",
|
||||
" \"key_ring\": key_ring,\n",
|
||||
" }\n",
|
||||
" )\n",
|
||||
" print(\"Created key ring: {}\".format(created_key_ring.name))\n",
|
||||
" return created_key_ring\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"key_ring = create_key_ring(\n",
|
||||
" project_id=PROJECT_ID, location=REGION, key_ring_id=KEY_RING_ID\n",
|
||||
")\n",
|
||||
"print(key_ring)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "gCL1-IfFtWXl"
|
||||
},
|
||||
"source": [
|
||||
"### Create a key\n",
|
||||
"\n",
|
||||
"Next, you create your key. Use the helper function `create_key()` with the following parameters:\n",
|
||||
"\n",
|
||||
"- `project_id`: Your project ID.\n",
|
||||
"- `location`: Your region.\n",
|
||||
"- `key_ring_id`: The unique identifier for your key ring.\n",
|
||||
"- `key_id`: The unique identifier for your key.\n",
|
||||
"\n",
|
||||
"The helper function calls the KMS client method `create_cryto_key()` to create your key.\n",
|
||||
"\n",
|
||||
"Learn more about [](https://cloud.google.com/kms/docs/samples/kms-create-key-symmetric-encrypt-decrypt)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "LXcagdmSnYYW"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"KEY_ID = \"your_cmek_key_id\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_key(project_id, location, key_ring_id, key_id):\n",
|
||||
" \"\"\"\n",
|
||||
" Creates a new symmetric encryption/decryption key in Cloud KMS.\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" project_id (string): Google Cloud project ID (e.g. 'my-project').\n",
|
||||
" location (string): Cloud KMS location (e.g. 'us-east1').\n",
|
||||
" key_ring_id (string): ID of the Cloud KMS key ring (e.g. 'my-key-ring').\n",
|
||||
" key_id (string): ID of the key to create (e.g. 'my-symmetric-key').\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" CryptoKey: Cloud KMS key.\n",
|
||||
"\n",
|
||||
" \"\"\"\n",
|
||||
"\n",
|
||||
" # Create the client.\n",
|
||||
" client = kms.KeyManagementServiceClient()\n",
|
||||
"\n",
|
||||
" # Build the parent key ring name.\n",
|
||||
" key_ring_name = client.key_ring_path(project_id, location, key_ring_id)\n",
|
||||
"\n",
|
||||
" # Build the key.\n",
|
||||
" purpose = kms.CryptoKey.CryptoKeyPurpose.ENCRYPT_DECRYPT\n",
|
||||
" algorithm = (\n",
|
||||
" kms.CryptoKeyVersion.CryptoKeyVersionAlgorithm.GOOGLE_SYMMETRIC_ENCRYPTION\n",
|
||||
" )\n",
|
||||
" key = {\n",
|
||||
" \"purpose\": purpose,\n",
|
||||
" \"version_template\": {\n",
|
||||
" \"algorithm\": algorithm,\n",
|
||||
" },\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Call the API.\n",
|
||||
" created_key = client.create_crypto_key(\n",
|
||||
" request={\"parent\": key_ring_name, \"crypto_key_id\": key_id, \"crypto_key\": key}\n",
|
||||
" )\n",
|
||||
" print(\"Created symmetric key: {}\".format(created_key.name))\n",
|
||||
" return created_key\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"key_id = create_key(\n",
|
||||
" project_id=PROJECT_ID, location=REGION, key_ring_id=KEY_RING_ID, key_id=KEY_ID\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(key_id)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "3gKDBOqC8Gl5"
|
||||
},
|
||||
"source": [
|
||||
"### Set service account permissions\n",
|
||||
"\n",
|
||||
"Next, you set permissions for your Vertex AI service account to encrypt and decrypt resources using your key.\n",
|
||||
"\n",
|
||||
"Learn more about [Grant Vertex AI permissions](https://cloud.google.com/vertex-ai/docs/general/cmek#grant_permissions)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "6QrRg08Vqfru"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Reference: https://cloud.google.com/vertex-ai/docs/general/cmek#granting_permissions\n",
|
||||
"# Get the service account\n",
|
||||
"SERVICE_ACCOUNT = ! gcloud projects get-iam-policy {PROJECT_ID} \\\n",
|
||||
" --flatten=\"bindings[].members\" \\\n",
|
||||
" --format=\"table(bindings.members)\" \\\n",
|
||||
" --filter=\"bindings.role:roles/aiplatform.serviceAgent\" \\\n",
|
||||
" | grep -oP \"service-.+?@gcp-sa-aiplatform.iam.gserviceaccount.com\"\n",
|
||||
"SERVICE_ACCOUNT = SERVICE_ACCOUNT[0]\n",
|
||||
"\n",
|
||||
"print(f\"Service account is: {SERVICE_ACCOUNT}\")\n",
|
||||
"\n",
|
||||
"# Give permissions\n",
|
||||
"! gcloud kms keys add-iam-policy-binding {KEY_ID} \\\n",
|
||||
" --keyring={KEY_RING_ID} \\\n",
|
||||
" --location={REGION} \\\n",
|
||||
" --project={PROJECT_ID} \\\n",
|
||||
" --member=serviceAccount:{SERVICE_ACCOUNT} \\\n",
|
||||
" --role=roles/cloudkms.cryptoKeyEncrypterDecrypter"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "1e8cd37e5f99"
|
||||
},
|
||||
"source": [
|
||||
"Create the full resource identifier for the created key"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ebAHZg2vlhXL"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"ENCRYPTION_SPEC_KEY_NAME = key_id.name"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "Aa_8wrqSkamz"
|
||||
},
|
||||
"source": [
|
||||
"## Initialize Vertex SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the *client* for Vertex AI\n",
|
||||
"\n",
|
||||
"All resources created during this Notebook run will encrypted with the encryption key created above.\n",
|
||||
"\n",
|
||||
"You can override the encryption key at each function call."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "init_aip:mbsdk,all"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI SDK for Python\n",
|
||||
"\n",
|
||||
"Initialize the Vertex AI SDK for Python for your project, bucket, and corresponding encryption key.\n",
|
||||
"\n",
|
||||
"All resources created during this session are encrypted with the encryption key you created.\n",
|
||||
"\n",
|
||||
"*Note:* You can override the encryption key at each function call."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "ohdgOs69kGNU"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"aip.init(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" staging_bucket=BUCKET_NAME,\n",
|
||||
" location=REGION,\n",
|
||||
" encryption_spec_key_name=ENCRYPTION_SPEC_KEY_NAME,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "import_file:u_dataset,csv"
|
||||
},
|
||||
"source": [
|
||||
"#### Location of Cloud Storage training data.\n",
|
||||
"\n",
|
||||
"Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "import_file:flowers,csv,icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"IMPORT_FILE = (\n",
|
||||
" \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "35QVNhACqcTJ"
|
||||
},
|
||||
"source": [
|
||||
"# Create `Vertex AI ImageDataset` resource\n",
|
||||
"\n",
|
||||
"Next, you create an `ImageDataset` resource, which will be encrypted using your encryption key."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "4OfCqaYRqcTJ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"dataset = aip.ImageDataset.create(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" gcs_source=[IMPORT_FILE],\n",
|
||||
" import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(dataset.resource_name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "6-bBqipfqcTS"
|
||||
},
|
||||
"source": [
|
||||
"# Launch a Training Job to Create a Model\n",
|
||||
"\n",
|
||||
"Train an AutoML Image Classification model."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "aA41rT_mb-rV"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"job = aiplatform.AutoMLImageTrainingJob(\n",
|
||||
" display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" multi_label=False,\n",
|
||||
" model_type=\"CLOUD\",\n",
|
||||
" base_model=None,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# This will take around half an hour to run\n",
|
||||
"model = job.run(\n",
|
||||
" dataset=ds,\n",
|
||||
" model_display_name=\"flowers_\" + TIMESTAMP,\n",
|
||||
" training_fraction_split=0.6,\n",
|
||||
" validation_fraction_split=0.2,\n",
|
||||
" test_fraction_split=0.2,\n",
|
||||
" budget_milli_node_hours=8000,\n",
|
||||
" disable_early_stopping=False,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "5vhDsMJNqcTW"
|
||||
},
|
||||
"source": [
|
||||
"# Deploy Your Model\n",
|
||||
"\n",
|
||||
"Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Y9GH72wWqcTX"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint = model.deploy()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "nIw1ifPuqcTb"
|
||||
},
|
||||
"source": [
|
||||
"# Predict on Endpoint\n",
|
||||
"- Take one sample from the data imported to the dataset\n",
|
||||
"- This sample will be encoded to base64 and passed to the endpoint for prediction"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "H23ISHdHVIZM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_item = !gsutil cat $IMPORT_FILE | head -n1\n",
|
||||
"test_item, test_label = str(test_item[0]).split(\",\")\n",
|
||||
"\n",
|
||||
"print(test_item, test_label)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "TF_N0kqZU768"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import base64\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(test_item, \"rb\") as f:\n",
|
||||
" content = f.read()\n",
|
||||
"\n",
|
||||
"# The format of each instance should conform to the deployed model's prediction input schema.\n",
|
||||
"instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n",
|
||||
"\n",
|
||||
"prediction = endpoint.predict(instances=instances_list)\n",
|
||||
"\n",
|
||||
"print(prediction)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "nWA3qocXfk82"
|
||||
},
|
||||
"source": [
|
||||
"# Undeploy Model from Endpoint"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "V1brMaO_fk82"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"endpoint.undeploy_all()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "e00750837ca8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# missing\n",
|
||||
"endpoint.delete()\n",
|
||||
"model.delete()\n",
|
||||
"dataset.delete()\n",
|
||||
"\n",
|
||||
"! gcloud kms keys versions destroy key-version \\\n",
|
||||
" --key key {KEY_ID} \\\n",
|
||||
" --keyring={KEY_RING_ID} \\\n",
|
||||
" --location={REGION} "
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "aa95b7fff9b5"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"! gcloud kms keys list --location {REGION} --keyring {KEY_RING_ID}"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"collapsed_sections": [],
|
||||
"name": "get_started_with_cmek_training.ipynb",
|
||||
"toc_visible": true
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/mlops_experimentation.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/mlops_experimentation.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage2/mlops_experimentation.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/mlops_experimentation.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -103,10 +103,10 @@
|
||||
"- Add a serving function for online/batch prediction to the custom model.\n",
|
||||
"- Test the custom model with the serving function.\n",
|
||||
"- Evaluate the custom model using Vertex AI Batch Prediction\n",
|
||||
"- Wait for AutoML training job to complete.\n",
|
||||
"- Wait for the AutoML training job to complete.\n",
|
||||
"- Evaluate the AutoML model using Vertex AI Batch Prediction with the same evaluation slices as the custom model.\n",
|
||||
"- Set the evaluation results of the AutoML model as the baseline.\n",
|
||||
"- If the evaluation of the custom model is below baseline, continue to experiment with custom model.\n",
|
||||
"- If the evaluation of the custom model is below baseline, continue to experiment with the custom model.\n",
|
||||
"- If the evaluation of the custom model is above baseline, save the model as the first best model."
|
||||
]
|
||||
},
|
||||
@@ -186,7 +186,9 @@
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -436,7 +438,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -783,12 +785,15 @@
|
||||
"source": [
|
||||
"import json\n",
|
||||
"\n",
|
||||
"with tf.io.gfile.GFile(\n",
|
||||
" \"gs://\" + dataset.labels[\"user_metadata\"] + \"/metadata.jsonl\", \"r\"\n",
|
||||
") as f:\n",
|
||||
" metadata = json.load(f)\n",
|
||||
"try:\n",
|
||||
" with tf.io.gfile.GFile(\n",
|
||||
" \"gs://\" + dataset.labels[\"user_metadata\"] + \"/metadata.jsonl\", \"r\"\n",
|
||||
" ) as f:\n",
|
||||
" metadata = json.load(f)\n",
|
||||
"\n",
|
||||
"print(metadata)"
|
||||
" print(metadata)\n",
|
||||
"except:\n",
|
||||
" print(\"no metadata\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1164,7 +1169,7 @@
|
||||
" display_name=\"chicago_\" + TIMESTAMP,\n",
|
||||
" artifact_uri=MODEL_DIR,\n",
|
||||
" serving_container_image_uri=DEPLOY_IMAGE,\n",
|
||||
" labels={\"base_model\": 1},\n",
|
||||
" labels={\"base_model\": \"1\"},\n",
|
||||
" sync=True,\n",
|
||||
")"
|
||||
]
|
||||
@@ -1213,7 +1218,7 @@
|
||||
"setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n",
|
||||
"! echo \"$setup_cfg\" > custom/setup.cfg\n",
|
||||
"\n",
|
||||
"setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'google-cloud-aiplatform',\\n\\n 'cloudml-hypertune',\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n",
|
||||
"setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'google-cloud-aiplatform',\\n\\n 'cloudml-hypertune',\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n 'tensorflow_data_validation==1.2',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n",
|
||||
"! echo \"$setup_py\" > custom/setup.py\n",
|
||||
"\n",
|
||||
"pkg_info = \"Metadata-Version: 1.0\\n\\nName: Chicago Taxi tabular binary classifier\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: cdpe@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex AI\"\n",
|
||||
@@ -1382,6 +1387,7 @@
|
||||
"from hypertune import HyperTune\n",
|
||||
"\n",
|
||||
"def compile(model, hyperparams):\n",
|
||||
" ''' Compile the model '''\n",
|
||||
" optimizer = tf.keras.optimizers.Adam(learning_rate=hyperparams[\"learning_rate\"])\n",
|
||||
" loss = tf.keras.losses.BinaryCrossentropy(from_logits=False)\n",
|
||||
" metrics = [tf.keras.metrics.BinaryAccuracy(name=\"accuracy\")]\n",
|
||||
@@ -1389,6 +1395,43 @@
|
||||
" model.compile(optimizer=optimizer,loss=loss, metrics=metrics)\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
"def warmup(\n",
|
||||
" model,\n",
|
||||
" hyperparams,\n",
|
||||
" train_data_dir,\n",
|
||||
" label_column,\n",
|
||||
" transformed_feature_spec\n",
|
||||
"):\n",
|
||||
" ''' Warmup the initialized model weights '''\n",
|
||||
"\n",
|
||||
" train_dataset = data.get_dataset(\n",
|
||||
" train_data_dir,\n",
|
||||
" transformed_feature_spec,\n",
|
||||
" label_column,\n",
|
||||
" batch_size=hyperparams[\"batch_size\"],\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" lr_inc = (hyperparams['end_learning_rate'] - hyperparams['start_learning_rate']) / hyperparams['num_epochs']\n",
|
||||
"\n",
|
||||
" def scheduler(epoch, lr):\n",
|
||||
" if epoch == 0:\n",
|
||||
" return hyperparams['start_learning_rate']\n",
|
||||
" return lr + lr_inc\n",
|
||||
"\n",
|
||||
"\n",
|
||||
" callbacks = [tf.keras.callbacks.LearningRateScheduler(scheduler)]\n",
|
||||
"\n",
|
||||
" logging.info(\"Model warmup started...\")\n",
|
||||
" history = model.fit(\n",
|
||||
" train_dataset,\n",
|
||||
" epochs=hyperparams[\"num_epochs\"],\n",
|
||||
" steps_per_epoch=hyperparams[\"steps\"],\n",
|
||||
" callbacks=callbacks\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" logging.info(\"Model warmup completed.\")\n",
|
||||
" return history\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def train(\n",
|
||||
" model,\n",
|
||||
@@ -1400,6 +1443,7 @@
|
||||
" log_dir,\n",
|
||||
" tuning=False\n",
|
||||
"):\n",
|
||||
" ''' Train the model '''\n",
|
||||
"\n",
|
||||
" train_dataset = data.get_dataset(\n",
|
||||
" train_data_dir,\n",
|
||||
@@ -1415,13 +1459,16 @@
|
||||
" batch_size=hyperparams[\"batch_size\"],\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" tensorboard = tf.keras.callbacks.TensorBoard(log_dir=log_dir)\n",
|
||||
"\n",
|
||||
" early_stop = tf.keras.callbacks.EarlyStopping(\n",
|
||||
" monitor=hyperparams[\"early_stop\"][\"monitor\"], patience=hyperparams[\"early_stop\"][\"patience\"], restore_best_weights=True\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" callbacks=[tensorboard, early_stop]\n",
|
||||
" callbacks = [early_stop]\n",
|
||||
"\n",
|
||||
" if log_dir:\n",
|
||||
" tensorboard = tf.keras.callbacks.TensorBoard(log_dir=log_dir)\n",
|
||||
"\n",
|
||||
" callbacks = callbacks.append(tensorboard)\n",
|
||||
"\n",
|
||||
" if tuning:\n",
|
||||
" # Instantiate the HyperTune reporting object\n",
|
||||
@@ -1437,6 +1484,8 @@
|
||||
" global_step=epoch\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" if not callbacks:\n",
|
||||
" callbacks = []\n",
|
||||
" callbacks.append(HPTCallback())\n",
|
||||
"\n",
|
||||
" logging.info(\"Model training started...\")\n",
|
||||
@@ -1483,6 +1532,7 @@
|
||||
"\n",
|
||||
"- `num_epochs`: The number of epochs to pass to the training package.\n",
|
||||
"- `compile()`: Compile the model for training.\n",
|
||||
"- `warmup()`: Warmup the initialized model weights.\n",
|
||||
"- `train()`: Train the model."
|
||||
]
|
||||
},
|
||||
@@ -1505,11 +1555,23 @@
|
||||
"logging.getLogger().setLevel(logging.INFO)\n",
|
||||
"\n",
|
||||
"hyperparams = {}\n",
|
||||
"hyperparams[\"learning_rate\"] = 0.001\n",
|
||||
"hyperparams[\"learning_rate\"] = 0.01\n",
|
||||
"aip.log_params(hyperparams)\n",
|
||||
"\n",
|
||||
"train.compile(model, hyperparams)\n",
|
||||
"\n",
|
||||
"warmupparams = {}\n",
|
||||
"warmupparams[\"start_learning_rate\"] = 0.0001\n",
|
||||
"warmupparams[\"end_learning_rate\"] = 0.01\n",
|
||||
"warmupparams[\"num_epochs\"] = 4\n",
|
||||
"warmupparams[\"batch_size\"] = 64\n",
|
||||
"warmupparams[\"steps\"] = 50\n",
|
||||
"aip.log_params(warmupparams)\n",
|
||||
"\n",
|
||||
"train.warmup(\n",
|
||||
" model, warmupparams, train_data_file_pattern, LABEL_COLUMN, transform_feature_spec\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"trainparams = {}\n",
|
||||
"trainparams[\"num_epochs\"] = 5\n",
|
||||
"trainparams[\"batch_size\"] = 64\n",
|
||||
@@ -1627,6 +1689,10 @@
|
||||
" - Compiles the model.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"- `warmup_model()`:\n",
|
||||
" - Warms up the initialized model weights\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"- `train_model()`:\n",
|
||||
" - Train the model.\n",
|
||||
"\n",
|
||||
@@ -1659,7 +1725,10 @@
|
||||
"from trainer import data\n",
|
||||
"from trainer import model as model_\n",
|
||||
"from trainer import train\n",
|
||||
"\n",
|
||||
"try:\n",
|
||||
" from trainer import serving\n",
|
||||
"except:\n",
|
||||
" pass\n",
|
||||
"\n",
|
||||
"parser = argparse.ArgumentParser()\n",
|
||||
"parser.add_argument('--model-dir', dest='model_dir',\n",
|
||||
@@ -1671,6 +1740,9 @@
|
||||
"parser.add_argument('--lr', dest='lr',\n",
|
||||
" default=0.001, type=float,\n",
|
||||
" help='Learning rate.')\n",
|
||||
"parser.add_argument('--start_lr', dest='start_lr',\n",
|
||||
" default=0.0001, type=float,\n",
|
||||
" help='Starting learning rate.')\n",
|
||||
"parser.add_argument('--epochs', dest='epochs',\n",
|
||||
" default=20, type=int,\n",
|
||||
" help='Number of epochs.')\n",
|
||||
@@ -1683,7 +1755,7 @@
|
||||
"parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n",
|
||||
" help='distributed training strategy')\n",
|
||||
"parser.add_argument('--tensorboard-log-dir', dest='tensorboard_log_dir',\n",
|
||||
" default='/tmp/logs', type=str,\n",
|
||||
" default=os.getenv('AIP_TENSORBOARD_LOG_DIR'), type=str,\n",
|
||||
" help='Output file for tensorboard logs')\n",
|
||||
"parser.add_argument('--experiment', dest='experiment',\n",
|
||||
" default=None, type=str,\n",
|
||||
@@ -1697,9 +1769,15 @@
|
||||
"parser.add_argument('--evaluate', dest='evaluate',\n",
|
||||
" default=False, type=bool,\n",
|
||||
" help='Whether to perform evaluation')\n",
|
||||
"parser.add_argument('--serving', dest='serving',\n",
|
||||
" default=False, type=bool,\n",
|
||||
" help='Whether to attach the serving function')\n",
|
||||
"parser.add_argument('--tuning', dest='tuning',\n",
|
||||
" default=False, type=bool,\n",
|
||||
" help='Whether to perform hyperparameter tuning')\n",
|
||||
"parser.add_argument('--warmup', dest='warmup',\n",
|
||||
" default=False, type=bool,\n",
|
||||
" help='Whether to perform warmup weight initialization')\n",
|
||||
"args = parser.parse_args()\n",
|
||||
"\n",
|
||||
"\n",
|
||||
@@ -1730,10 +1808,12 @@
|
||||
" aip.init(experiment=args.experiment, project=args.project)\n",
|
||||
" aip.start_run(args.run)\n",
|
||||
"\n",
|
||||
"metadata = {}\n",
|
||||
"\n",
|
||||
"def get_data():\n",
|
||||
" ''' Get the preprocessed training data '''\n",
|
||||
" global train_data_file_pattern, val_data_file_pattern, test_data_file_pattern\n",
|
||||
" global label_column, transform_feature_spec\n",
|
||||
" global label_column, transform_feature_spec, metadata\n",
|
||||
"\n",
|
||||
" dataset = aip.TabularDataset(args.dataset_id)\n",
|
||||
" METADATA = 'gs://' + dataset.labels['user_metadata'] + \"/metadata.jsonl\"\n",
|
||||
@@ -1754,6 +1834,8 @@
|
||||
"\n",
|
||||
"def get_model():\n",
|
||||
" ''' Get the untrained model architecture '''\n",
|
||||
" global model_artifacts\n",
|
||||
"\n",
|
||||
" vertex_model = model_.get(args.model_id)\n",
|
||||
" model_artifacts = vertex_model.gca_resource.artifact_uri\n",
|
||||
" model = tf.keras.models.load_model(model_artifacts)\n",
|
||||
@@ -1764,9 +1846,25 @@
|
||||
" if args.experiment:\n",
|
||||
" aip.log_params(hyperparams)\n",
|
||||
"\n",
|
||||
" metadata.update(hyperparams)\n",
|
||||
" with tf.io.gfile.GFile(os.path.join(args.model_dir, \"metrics.txt\"), \"w\") as f:\n",
|
||||
" f.write(json.dumps(metadata))\n",
|
||||
"\n",
|
||||
" train.compile(model, hyperparams)\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
"def warmup_model(model):\n",
|
||||
" ''' Warmup the initialized model weights '''\n",
|
||||
" warmupparams = {}\n",
|
||||
" warmupparams[\"num_epochs\"] = args.epochs\n",
|
||||
" warmupparams[\"batch_size\"] = args.batch_size\n",
|
||||
" warmupparams[\"steps\"] = args.steps\n",
|
||||
" warmupparams[\"start_learning_rate\"] = args.start_lr\n",
|
||||
" warmupparams[\"end_learning_rate\"] = args.lr\n",
|
||||
"\n",
|
||||
" train.warmup(model, warmupparams, train_data_file_pattern, label_column, transform_feature_spec)\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
"def train_model(model):\n",
|
||||
" ''' Train the model '''\n",
|
||||
" trainparams = {}\n",
|
||||
@@ -1775,6 +1873,11 @@
|
||||
" trainparams[\"early_stop\"] = {\"monitor\": \"val_loss\", \"patience\": 5}\n",
|
||||
" if args.experiment:\n",
|
||||
" aip.log_params(trainparams)\n",
|
||||
"\n",
|
||||
" metadata.update(trainparams)\n",
|
||||
" with tf.io.gfile.GFile(os.path.join(args.model_dir, \"metrics.txt\"), \"w\") as f:\n",
|
||||
" f.write(json.dumps(metadata))\n",
|
||||
"\n",
|
||||
" train.train(model, trainparams, train_data_file_pattern, val_data_file_pattern, label_column, transform_feature_spec, args.tensorboard_log_dir, args.tuning)\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
@@ -1783,19 +1886,36 @@
|
||||
" evalparams = {}\n",
|
||||
" evalparams[\"batch_size\"] = args.batch_size\n",
|
||||
" metrics = train.evaluate(model, evalparams, test_data_file_pattern, label_column, transform_feature_spec)\n",
|
||||
" with tf.io.gfile.GFile(os.path.join(args.model_dir, \"metrics.txt\", \"w\")) as f:\n",
|
||||
" f.write(str(metrics))\n",
|
||||
"\n",
|
||||
" metadata.update({'metrics': metrics})\n",
|
||||
" with tf.io.gfile.GFile(os.path.join(args.model_dir, \"metrics.txt\"), \"w\") as f:\n",
|
||||
" f.write(json.dumps(metadata))\n",
|
||||
"\n",
|
||||
"get_data()\n",
|
||||
"with strategy.scope():\n",
|
||||
" model = get_model()\n",
|
||||
"model = train_model(model)\n",
|
||||
"\n",
|
||||
"if args.warmup:\n",
|
||||
" model = warmup_model(model)\n",
|
||||
"else:\n",
|
||||
" model = train_model(model)\n",
|
||||
"\n",
|
||||
"if args.evaluate:\n",
|
||||
" evaluate_model(model)\n",
|
||||
"\n",
|
||||
"logging.info('Save trained model to: ' + args.model_dir)\n",
|
||||
"model.save(args.model_dir)"
|
||||
"if args.serving:\n",
|
||||
" logging.info('Save serving model to: ' + args.model_dir)\n",
|
||||
" serving.construct_serving_model(\n",
|
||||
" model=model,\n",
|
||||
" serving_model_dir=args.model_dir,\n",
|
||||
" metadata=metadata\n",
|
||||
" )\n",
|
||||
"elif args.warmup:\n",
|
||||
" logging.info('Save warmed up model to: ' + model_artifacts)\n",
|
||||
" model.save(model_artifacts)\n",
|
||||
"else:\n",
|
||||
" logging.info('Save trained model to: ' + args.model_dir)\n",
|
||||
" model.save(args.model_dir)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1822,6 +1942,30 @@
|
||||
"!cd custom; python3 -m trainer.task --model-id={MODEL_ID} --dataset-id={DATASET_ID} --experiment='chicago' --run='test' --project={PROJECT_ID} --epochs=5 --model-dir=/tmp --evaluate=True"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "warmup_base_model"
|
||||
},
|
||||
"source": [
|
||||
"### Warmup training\n",
|
||||
"\n",
|
||||
"Now that you have tested the training scripts, you perform warmup training on the base model. Warmup training is used to stabilize the weight initialization. By doing so, each subsequent training and tuning of the model architecture will start with the same stabilized weight initialization."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "warmup_base_model"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"MODEL_DIR = f\"{BUCKET_NAME}/base_model\"\n",
|
||||
"\n",
|
||||
"!cd custom; python3 -m trainer.task --model-id={MODEL_ID} --dataset-id={DATASET_ID} --project={PROJECT_ID} --epochs=5 --steps=300 --batch_size=16 --lr=0.01 --start_lr=0.0001 --model-dir={MODEL_DIR} --warmup=True"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -1966,6 +2110,7 @@
|
||||
" accelerator_count=TRAIN_NGPU,\n",
|
||||
" base_output_dir=MODEL_DIR,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" tensorboard=tensorboard_resource_name,\n",
|
||||
" sync=True,\n",
|
||||
")"
|
||||
]
|
||||
@@ -2404,6 +2549,7 @@
|
||||
" accelerator_count=TRAIN_NGPU,\n",
|
||||
" base_output_dir=MODEL_DIR,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" tensorboard=tensorboard_resource_name,\n",
|
||||
" sync=True,\n",
|
||||
")"
|
||||
]
|
||||
@@ -2449,8 +2595,10 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"EXPERIMENT_NAME = \"chicago\"\n",
|
||||
"\n",
|
||||
"experiment_df = aip.get_experiment_df()\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == \"chicago\"]\n",
|
||||
"experiment_df = experiment_df[experiment_df.experiment_name == EXPERIMENT_NAME]\n",
|
||||
"experiment_df.T"
|
||||
]
|
||||
},
|
||||
@@ -2477,6 +2625,28 @@
|
||||
"! gsutil cat $METRICS"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_tensorboard"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the TensorBoard instance\n",
|
||||
"\n",
|
||||
"Next, delete the TensorBoard instance."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_tensorboard"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"tensorboard.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -2508,6 +2678,13 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile custom/trainer/serving.py\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"import tensorflow_data_validation as tfdv\n",
|
||||
"import tensorflow_transform as tft\n",
|
||||
"import logging\n",
|
||||
"\n",
|
||||
"def _get_serve_features_fn(model, tft_output):\n",
|
||||
" \"\"\"Returns a function that accept a dictionary of features and applies TFT.\"\"\"\n",
|
||||
"\n",
|
||||
@@ -2521,8 +2698,8 @@
|
||||
" probabilities = model(transformed_features)\n",
|
||||
" return {\"scores\": probabilities}\n",
|
||||
"\n",
|
||||
" return serve_features_fn\n",
|
||||
"\n",
|
||||
" return serve_features_fn\n",
|
||||
"\n",
|
||||
"def _get_serve_tf_examples_fn(model, tft_output, feature_spec):\n",
|
||||
" \"\"\"Returns a function that parses a serialized tf.Example and applies TFT.\"\"\"\n",
|
||||
@@ -2544,23 +2721,18 @@
|
||||
"\n",
|
||||
" return serve_tf_examples_fn\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def construct_serving_model(model, serving_model_dir, metadata):\n",
|
||||
"def construct_serving_model(\n",
|
||||
" model, serving_model_dir, metadata\n",
|
||||
"):\n",
|
||||
" global features\n",
|
||||
"\n",
|
||||
" schema_location = metadata[\"schema\"]\n",
|
||||
" features = (\n",
|
||||
" metadata[\"numeric_features\"]\n",
|
||||
" + metadata[\"categorical_features\"]\n",
|
||||
" + metadata[\"embedding_features\"]\n",
|
||||
" )\n",
|
||||
" schema_location = metadata['schema']\n",
|
||||
" features = metadata['numeric_features'] + metadata['categorical_features'] + metadata['embedding_features']\n",
|
||||
" print(\"FEATURES\", features)\n",
|
||||
" tft_output_dir = metadata[\"transform_artifacts_dir\"]\n",
|
||||
"\n",
|
||||
" schema = tfdv.load_schema_text(schema_location)\n",
|
||||
" feature_spec = tft.tf_metadata.schema_utils.schema_as_feature_spec(\n",
|
||||
" schema\n",
|
||||
" ).feature_spec\n",
|
||||
" feature_spec = tft.tf_metadata.schema_utils.schema_as_feature_spec(schema).feature_spec\n",
|
||||
"\n",
|
||||
" tft_output = tft.TFTransformOutput(tft_output_dir)\n",
|
||||
"\n",
|
||||
@@ -2584,9 +2756,9 @@
|
||||
" ),\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" logging.info(\"Model export started...\")\n",
|
||||
" logging.info(\"Model saving started...\")\n",
|
||||
" model.save(serving_model_dir, signatures=signatures)\n",
|
||||
" logging.info(\"Model export completed.\")"
|
||||
" logging.info(\"Model saving completed.\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -2608,13 +2780,19 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"os.chdir(\"custom\")\n",
|
||||
"\n",
|
||||
"from trainer import serving\n",
|
||||
"\n",
|
||||
"SERVING_MODEL_DIR = BUCKET_NAME + \"/serving_model\"\n",
|
||||
"\n",
|
||||
"construct_serving_model(\n",
|
||||
"serving.construct_serving_model(\n",
|
||||
" model=model, serving_model_dir=SERVING_MODEL_DIR, metadata=metadata\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"serving_model = tf.keras.models.load_model(SERVING_MODEL_DIR)"
|
||||
"serving_model = tf.keras.models.load_model(SERVING_MODEL_DIR)\n",
|
||||
"\n",
|
||||
"os.chdir(\"..\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -35,15 +35,120 @@ The third stage in MLOps is formalization to develop an automated pipeline proce
|
||||
|
||||
[Get Started with Kubeflow pipelines](get_started_with_kubeflow_pipelines.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Building KFP lightweight Python function components.
|
||||
- Assembling and compiling KFP components into a pipeline.
|
||||
- Executing a KFP pipeline using Vertex AI Pipelines.
|
||||
- Loading component and pipeline definitions from a source code repository.
|
||||
- Building sequential, parallel, multiple output components.
|
||||
- Building control flow into pipelines.
|
||||
```
|
||||
|
||||
[Get Started with BQ and TFDV components](get_started_with_bq_tfdv_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Build and execute a pipeline component for creating a Vertex AI Tabular Dataset from a BigQuery table.
|
||||
- Build and execute a pipeline component for generating TFDV statistics and schema from a Vertex AI Tabular Dataset.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Dataflow components](get_started_with_dataflow_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Build an Apache Beam data pipeline.
|
||||
- Encapsulate the Apache Beam data pipeline with a Dataflow component in a Vertex AI pipeline.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI AutoML components](get_started_with_automl_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Training a Vertex AI AutoML trained model.
|
||||
- Test the serving binary with a batch prediction job.
|
||||
- Deploying a Vertex AI AutoML trained model.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI Custom Training components](get_started_with_custom_training_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Training a Vertex AI custom trained model.
|
||||
- Test the serving binary with a batch prediction job.
|
||||
- Deploying a Vertex AI custom trained model.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with Vertex AI Hyperparameter Tuning components](get_started_with_hpt_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Hyperparameter tune/train a custom model.
|
||||
- Retrieve the tuned hyperparameter values and metrics to optimize.
|
||||
- If the metrics exceed a specified threshold.
|
||||
- Get the location of the model artifacts for the best tuned model.
|
||||
- Upload the model artifacts to a `Vertex AI Model` resource.
|
||||
- Execute a Vertex AI pipeline.
|
||||
```
|
||||
|
||||
[Get Started with BQML components](get_started_with_bqml_pipeline_components.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Construct a pipeline for:
|
||||
- Training BigQuery ML model.
|
||||
- Evaluating the BigQuery ML model.
|
||||
- Exporting the BigQuery ML model.
|
||||
- Importing the BigQuery ML model to a Vertex AI model.
|
||||
- Deploy the Vertex AI model.
|
||||
- Execute a Vertex AI pipeline.
|
||||
- Make a prediction with the deployed Vertex AI model.
|
||||
```
|
||||
|
||||
[Get Started with rapid prototyping with BQML and AutoML components](get_started_with_rapid_prototyping_bqml_automl.ipynb)
|
||||
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Creating a BigQuery and Vertex AI training dataset.
|
||||
- Training a BigQuery ML and AutoML model.
|
||||
- Extracting evaluation metrics from the BigQueryML and AutoML models.
|
||||
- Selecting the best trained model.
|
||||
- Deploying the best trained model.
|
||||
- Testing the deployed model infrastructure.
|
||||
```
|
||||
|
||||
### E2E Stage Example
|
||||
|
||||
[Stage 3: Formalization](mlops_formalization.ipynb)
|
||||
|
||||
```
|
||||
The steps performed include:
|
||||
|
||||
- Obtain resources from the experimentation stage.
|
||||
- Baseline model.
|
||||
- Dataset schema/statistics for baseline model.
|
||||
- Formalize a data preprocessing pipeline.
|
||||
- Extract columns/rows from BigQuery table to local BigQuery table.
|
||||
- Use Tensorflow Data Validation library to determine statistics, schema, and features.
|
||||
- Use Dataflow to preprocess the data.
|
||||
- Create a Vertex AI Dataset.
|
||||
- Formalize a build model architecture pipeline.
|
||||
- Create the Vertex AI Model base model.
|
||||
- Formalize a training pipeline.
|
||||
```
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_automl_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -86,10 +86,14 @@
|
||||
"- `Vertex AI AutoML`\n",
|
||||
"- `Google Cloud Pipeline Components`\n",
|
||||
"- `Vertex AI Dataset, Model and Endpoint` resources\n",
|
||||
"- `Vertex AI Prediction`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Construct a pipeline for training and deploying a Vertex AI AutoML model.\n",
|
||||
"- Construct a pipeline for:\n",
|
||||
" - Training a Vertex AI AutoML trained model.\n",
|
||||
" - Test the serving binary with a batch prediction job.\n",
|
||||
" - Deploying a Vertex AI AutoML trained model.\n",
|
||||
"- Execute a Vertex AI pipeline."
|
||||
]
|
||||
},
|
||||
@@ -119,12 +123,17 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade python-tabulate $USER_FLAG\n",
|
||||
" ! pip3 install -U opencv-python-headless==4.5.2.52 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -221,7 +230,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -374,7 +383,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -518,8 +527,8 @@
|
||||
"\n",
|
||||
"The Vertex AI pre-built pipeline components does not currently have a component for retrieiving the model evaluations for a AutoML model. So, you will first write your own component, as follows:\n",
|
||||
"\n",
|
||||
"- Takes as input the project, region and Model artifacts returned from an AutoML training component.\n",
|
||||
"- Create a client interface to the Vertex AI Model service.\n",
|
||||
"- Takes as input the region and Model artifacts returned from an AutoML training component.\n",
|
||||
"- Create a client interface to the Vertex AI Model service (`metadata[\"resource_name\"]).\n",
|
||||
"- Construct the resource ID for the model from the model artifact parameter.\n",
|
||||
"- Retrieve the model evaluation\n",
|
||||
"- Return the model evaluation as a string."
|
||||
@@ -533,11 +542,13 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from kfp.v2.dsl import Artifact, Input, Model\n",
|
||||
"from kfp.v2.dsl import Artifact, Input, Model, Output\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@component(packages_to_install=[\"google-cloud-aiplatform\"])\n",
|
||||
"def evaluateAutoMLModelOp(model: Input[Artifact], region: str) -> str:\n",
|
||||
"def evaluateAutoMLModelOp(\n",
|
||||
" model: Input[Artifact], region: str, model_evaluation: Output[Artifact]\n",
|
||||
"):\n",
|
||||
" import logging\n",
|
||||
"\n",
|
||||
" import google.cloud.aiplatform.gapic as gapic\n",
|
||||
@@ -550,17 +561,16 @@
|
||||
"\n",
|
||||
" model_evaluations = model_service_client.list_model_evaluations(parent=model_id)\n",
|
||||
" model_evaluation = list(model_evaluations)[0]\n",
|
||||
" logging.info(model_evaluation)\n",
|
||||
" return str(model_evaluation)"
|
||||
" logging.info(model_evaluation)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_automl_pipeline:automl,icn"
|
||||
"id": "create_automl_pipeline:icn"
|
||||
},
|
||||
"source": [
|
||||
"## Construct AutoML Training Pipeline\n",
|
||||
"## Construct AutoML training pipeline\n",
|
||||
"\n",
|
||||
"In the example below, you construct a pipeline for training an AutoML model using pre-built Google Cloud Pipeline Components for AutoML, as follows:\n",
|
||||
"\n",
|
||||
@@ -568,13 +578,25 @@
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The import file for the dataset is passed into the pipeline.\n",
|
||||
" - The component returns the dataset resource as `outputs[\"dataset\"]`\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"2. Use the prebuilt component `AutoMLImageTrainingJobRunOp` to train a Vertex AI AutoML Model resource, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The dataset is the output from the `ImageDatasetCreateOp`.\n",
|
||||
"3. Use the prebuilt component `EndpointCreateOp` to create a Vertex AI Endpoint to deploy the trained model to, where:\n",
|
||||
" - The component returns the model resource as `outputs[\"model\"]`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"3. Use the prebuild component `ModelBatchPredictOp` to do a test batch prediction, where:\n",
|
||||
" - The model is the output from the `AutoMLTrainingJobRunOp`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"4. Use the prebuilt component `EndpointCreateOp` to create a Vertex AI Endpoint to deploy the trained model to, where:\n",
|
||||
" - Since the component has no dependencies on other components, by default it would be executed in parallel with the model training.\n",
|
||||
" - The `after(training_op)` is added to serialize its execution, so its only executed if the training operation completes successfully.\n",
|
||||
"4. Use the prebuilt component `ModelDeployOp` to deploy the trained AutoML model to, where:\n",
|
||||
" - The component returns the endpoint resource as `outputs[\"endpoint\"]`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"5. Use the prebuilt component `ModelDeployOp` to deploy the trained AutoML model to, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The model is the output from the `AutoMLTrainingJobRunOp`.\n",
|
||||
" - The endpoint is the output from the `EndpointCreateOp`\n",
|
||||
@@ -586,20 +608,26 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_automl_pipeline:automl,icn"
|
||||
"id": "create_automl_pipeline:icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/automl_icn_training\".format(BUCKET_NAME)\n",
|
||||
"DEPLOY_COMPUTE = \"n1-standard-4\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(\n",
|
||||
" name=\"automl-icn-training\", description=\"AutoML image classification training\"\n",
|
||||
")\n",
|
||||
"def pipeline(\n",
|
||||
" import_file: str, display_name: str, project: str = PROJECT_ID, region: str = REGION\n",
|
||||
" import_file: str,\n",
|
||||
" batch_files: list,\n",
|
||||
" display_name: str,\n",
|
||||
" bucket: str = PIPELINE_ROOT,\n",
|
||||
" project: str = PROJECT_ID,\n",
|
||||
" region: str = REGION,\n",
|
||||
"):\n",
|
||||
"\n",
|
||||
" dataset_op = gcc_aip.ImageDatasetCreateOp(\n",
|
||||
@@ -614,7 +642,6 @@
|
||||
" display_name=display_name,\n",
|
||||
" prediction_type=\"classification\",\n",
|
||||
" model_type=\"CLOUD\",\n",
|
||||
" base_model=None,\n",
|
||||
" dataset=dataset_op.outputs[\"dataset\"],\n",
|
||||
" model_display_name=display_name,\n",
|
||||
" training_fraction_split=0.6,\n",
|
||||
@@ -625,11 +652,25 @@
|
||||
"\n",
|
||||
" eval_op = evaluateAutoMLModelOp(model=training_op.outputs[\"model\"], region=region)\n",
|
||||
"\n",
|
||||
" batch_op = gcc_aip.ModelBatchPredictOp(\n",
|
||||
" project=project,\n",
|
||||
" job_display_name=\"batch_predict_job\",\n",
|
||||
" model=training_op.outputs[\"model\"],\n",
|
||||
" gcs_source_uris=batch_files,\n",
|
||||
" gcs_destination_output_uri_prefix=bucket,\n",
|
||||
" instances_format=\"jsonl\",\n",
|
||||
" predictions_format=\"jsonl\",\n",
|
||||
" model_parameters={},\n",
|
||||
" machine_type=DEPLOY_COMPUTE,\n",
|
||||
" starting_replica_count=1,\n",
|
||||
" max_replica_count=1,\n",
|
||||
" ).after(eval_op)\n",
|
||||
"\n",
|
||||
" endpoint_op = gcc_aip.EndpointCreateOp(\n",
|
||||
" project=project,\n",
|
||||
" location=region,\n",
|
||||
" display_name=display_name,\n",
|
||||
" ).after(eval_op)\n",
|
||||
" ).after(batch_op)\n",
|
||||
"\n",
|
||||
" deploy_op = gcc_aip.ModelDeployOp(\n",
|
||||
" model=training_op.outputs[\"model\"],\n",
|
||||
@@ -642,7 +683,108 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "run_automl_pipeline:automl,icn"
|
||||
"id": "get_test_items:batch_prediction"
|
||||
},
|
||||
"source": [
|
||||
"### Get test item(s)\n",
|
||||
"\n",
|
||||
"Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "get_test_items:automl,icn,csv"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n2\n",
|
||||
"if len(str(test_items[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
"else:\n",
|
||||
" test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
"\n",
|
||||
"print(test_item_1, test_label_1)\n",
|
||||
"print(test_item_2, test_label_2)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "copy_test_items:batch_prediction"
|
||||
},
|
||||
"source": [
|
||||
"### Copy test item(s)\n",
|
||||
"\n",
|
||||
"For the batch prediction, copy the test items over to your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "copy_test_items:batch_prediction"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"file_1 = test_item_1.split(\"/\")[-1]\n",
|
||||
"file_2 = test_item_2.split(\"/\")[-1]\n",
|
||||
"\n",
|
||||
"! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"\n",
|
||||
"test_item_1 = BUCKET_NAME + \"/\" + file_1\n",
|
||||
"test_item_2 = BUCKET_NAME + \"/\" + file_2"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "make_batch_file:automl,image"
|
||||
},
|
||||
"source": [
|
||||
"### Make the batch input file\n",
|
||||
"\n",
|
||||
"Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n",
|
||||
"\n",
|
||||
"- `content`: The Cloud Storage path to the image.\n",
|
||||
"- `mime_type`: The content type. In our example, it is a `jpeg` file.\n",
|
||||
"\n",
|
||||
"For example:\n",
|
||||
"\n",
|
||||
" {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "make_batch_file:automl,image"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"\n",
|
||||
"gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n",
|
||||
"with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n",
|
||||
" data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n",
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
" data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n",
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "run_automl_pipeline:icn"
|
||||
},
|
||||
"source": [
|
||||
"### Compile and execute the pipeline\n",
|
||||
@@ -650,6 +792,7 @@
|
||||
"Next, you compile the pipeline and then exeute it. The pipeline takes the following parameters, which are passed as the dictionary `parameter_values`:\n",
|
||||
"\n",
|
||||
"- `import_file`: The Cloud Storage path to the dataset index file.\n",
|
||||
"- `batch_files`: A list of Cloud Storage paths to the input batch files.\n",
|
||||
"- `display_name`: The display name for the generated Vertex AI resources.\n",
|
||||
"- `project`: The project ID.\n",
|
||||
"- `region`: The region."
|
||||
@@ -659,7 +802,7 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_automl_pipeline:automl,icn"
|
||||
"id": "run_automl_pipeline:icn"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -673,6 +816,7 @@
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" \"import_file\": IMPORT_FILE,\n",
|
||||
" \"batch_files\": [gcs_input_uri],\n",
|
||||
" \"display_name\": \"flowers\" + TIMESTAMP,\n",
|
||||
" \"project\": PROJECT_ID,\n",
|
||||
" \"region\": REGION,\n",
|
||||
@@ -724,27 +868,68 @@
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/executor_output.json\"\n",
|
||||
" )\n",
|
||||
" GCP_RESOURCES = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/gcp_resources\"\n",
|
||||
" )\n",
|
||||
" EVAL_METRICS = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/evaluation_metrics\"\n",
|
||||
" )\n",
|
||||
" if tf.io.gfile.exists(EXECUTE_OUTPUT):\n",
|
||||
" ! gsutil cat $EXECUTE_OUTPUT\n",
|
||||
" break\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
" elif tf.io.gfile.exists(GCP_RESOURCES):\n",
|
||||
" ! gsutil cat $GCP_RESOURCES\n",
|
||||
" return GCP_RESOURCES\n",
|
||||
" elif tf.io.gfile.exists(EVAL_METRICS):\n",
|
||||
" ! gsutil cat $EVAL_METRICS\n",
|
||||
" return EVAL_METRICS\n",
|
||||
"\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
" return None\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"print(\"imagedataset-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"imagedataset-create\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"automlimagetrainingjob-run\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"automlimagetrainingjob-run\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"image-dataset-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"image-dataset-create\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"automl-image-training-job\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"automl-image-training-job\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"endpoint-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"endpoint-create\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-deploy\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-deploy\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"evaluateautomlmodelop\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"evaluateautomlmodelop\")"
|
||||
"artifacts = print_pipeline_output(pipeline, \"evaluateautomlmodelop\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-batch-predict\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-batch-predict\")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\n",
|
||||
" output[\"artifacts\"][\"batchpredictionjob\"][\"artifacts\"][0][\"metadata\"][\n",
|
||||
" \"gcsOutputDirectory\"\n",
|
||||
" ]\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -821,6 +1006,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_bq_tfdv_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -119,6 +119,7 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
@@ -221,7 +222,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -670,9 +671,24 @@
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/executor_output.json\"\n",
|
||||
" )\n",
|
||||
" GCP_RESOURCES = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/gcp_resources\"\n",
|
||||
" )\n",
|
||||
" if tf.io.gfile.exists(EXECUTE_OUTPUT):\n",
|
||||
" ! gsutil cat $EXECUTE_OUTPUT\n",
|
||||
" break\n",
|
||||
" elif tf.io.gfile.exists(GCP_RESOURCES):\n",
|
||||
" ! gsutil cat $GCP_RESOURCES\n",
|
||||
" break\n",
|
||||
"\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
"\n",
|
||||
@@ -924,6 +940,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_custom_training_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_custom_training_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_custom_training_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_custom_training_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -86,10 +86,14 @@
|
||||
"- `Vertex AI Training`\n",
|
||||
"- `Google Cloud Pipeline Components`\n",
|
||||
"- `Vertex AI Dataset, Model and Endpoint` resources\n",
|
||||
"- `Vertex AI Prediction`\n",
|
||||
"\n",
|
||||
"The steps performed include:\n",
|
||||
"\n",
|
||||
"- Construct a pipeline for training and deploying a Vertex AI custom trained model.\n",
|
||||
"- Construct a pipeline for:\n",
|
||||
" - Training a Vertex AI custom trained model.\n",
|
||||
" - Test the serving binary with a batch prediction job.\n",
|
||||
" - Deploying a Vertex AI custom trained model.\n",
|
||||
"- Execute a Vertex AI pipeline."
|
||||
]
|
||||
},
|
||||
@@ -125,7 +129,11 @@
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade python-tabulate $USER_FLAG\n",
|
||||
" ! pip3 install -U opencv-python-headless==4.5.2.52 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -375,7 +383,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -768,6 +776,7 @@
|
||||
"import json\n",
|
||||
"import logging\n",
|
||||
"import tqdm\n",
|
||||
"import hypertune as hpt\n",
|
||||
"\n",
|
||||
"def parse_args():\n",
|
||||
" parser = argparse.ArgumentParser(description=\"TF.Keras Image Classification\")\n",
|
||||
@@ -794,7 +803,9 @@
|
||||
" parser.add_argument(\n",
|
||||
" \"--lr\", dest=\"lr\", default=0.01, type=float, help=\"Learning rate.\"\n",
|
||||
" )\n",
|
||||
" parser.add_argument(\"--batch-size\", default=16, type=int, help=\"mini-batch size\")\n",
|
||||
" parser.add_argument(\n",
|
||||
" \"--batch-size\", dest=\"batch_size\", default=16, type=int, help=\"mini-batch size\"\n",
|
||||
" )\n",
|
||||
" parser.add_argument(\n",
|
||||
" \"--epochs\", default=10, type=int, help=\"number of training epochs\"\n",
|
||||
" )\n",
|
||||
@@ -813,6 +824,14 @@
|
||||
" help=\"distributed training strategy\",\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" parser.add_argument(\n",
|
||||
" \"--tuning\",\n",
|
||||
" dest=\"tuning\",\n",
|
||||
" type=bool,\n",
|
||||
" default=False,\n",
|
||||
" help=\"hyperparameter tuning\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" args = parser.parse_args()\n",
|
||||
" return args\n",
|
||||
"\n",
|
||||
@@ -938,8 +957,19 @@
|
||||
"def train_model(model, train_dataset, val_dataset):\n",
|
||||
" logging.info(\"Start model training\")\n",
|
||||
" history = model.fit(\n",
|
||||
" x=train_dataset, epochs=args.epochs, validation_data=val_dataset, steps_per_epoch=args.steps\n",
|
||||
" x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps, batch_size=args.batch_size, validation_data=val_dataset\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" if args.tuning:\n",
|
||||
" hp_metric = history.history['val_accuracy'][-1]\n",
|
||||
"\n",
|
||||
" hpt = hypertune.HyperTune()\n",
|
||||
" hpt.report_hyperparameter_tuning_metric(\n",
|
||||
" hyperparameter_metric_tag='accuracy',\n",
|
||||
" metric_value=hp_metric,\n",
|
||||
" global_step=args.epochs\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" return history\n",
|
||||
"\n",
|
||||
"num_classes, train_dataset, val_dataset = get_data()\n",
|
||||
@@ -1006,21 +1036,32 @@
|
||||
" - `python_package`: The custom training Python package.\n",
|
||||
" - `python_module`: The entry module in the package to execute.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"2. Use the prebuilt component `ImageDatasetCreateOp` to create a Vertex AI Dataset resource, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The import file for the dataset is passed into the pipeline.\n",
|
||||
" - The component returns the dataset resource as `outputs[\"dataset\"]`\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"3. Use the prebuilt component `CustomPythonPackageTrainingJobRunOp` to train a custom model and upload the custom model as a Vertex AI Model resource, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The dataset is the output from the `ImageDatasetCreateOp`.\n",
|
||||
" - The python package, command line argument are passed into the pipeline.\n",
|
||||
" - The training and serving containers are specified in the pipeline definition.\n",
|
||||
" - The component returns the model resource as `outputs[\"model\"]`.\n",
|
||||
"4. Use the prebuilt component `EndpointCreateOp` to create a Vertex AI Endpoint to deploy the trained model to, where:\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"4. Use the prebuild component `ModelBatchPredictOp` to do a test batch prediction, where:\n",
|
||||
" - The model is the output from the `CustomPythonPackageTrainingJobRunOp`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"5. Use the prebuilt component `EndpointCreateOp` to create a Vertex AI Endpoint to deploy the trained model to, where:\n",
|
||||
" - Since the component has no dependencies on other components, by default it would be executed in parallel with the model training.\n",
|
||||
" - The `after(training_op)` is added to serialize its execution, so its only executed if the training operation completes successfully.\n",
|
||||
" - The component returns the endpoint resource as `outputs[\"endpoint\"]`.\n",
|
||||
"5. Use the prebuilt component `ModelDeployOp` to deploy the trained Vertex AI model to, where:\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"6. Use the prebuilt component `ModelDeployOp` to deploy the trained Vertex AI model to, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The model is the output from the `CustomPythonPackageTrainingJobRunOp`.\n",
|
||||
" - The endpoint is the output from the `EndpointCreateOp`\n",
|
||||
@@ -1036,9 +1077,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/custom_icn_training\".format(BUCKET_NAME)\n",
|
||||
"DEPLOY_COMPUTE = \"n1-standard-4\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(\n",
|
||||
@@ -1048,11 +1088,18 @@
|
||||
"def pipeline(\n",
|
||||
" import_file: str,\n",
|
||||
" display_name: str,\n",
|
||||
" batch_files: list,\n",
|
||||
" python_package: str,\n",
|
||||
" python_module: str,\n",
|
||||
" bucket: str = PIPELINE_ROOT,\n",
|
||||
" project: str = PROJECT_ID,\n",
|
||||
" region: str = REGION,\n",
|
||||
"):\n",
|
||||
" from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
" from google_cloud_pipeline_components.v1.batch_predict_job import \\\n",
|
||||
" ModelBatchPredictOp\n",
|
||||
" from google_cloud_pipeline_components.v1.endpoint import (EndpointCreateOp,\n",
|
||||
" ModelDeployOp)\n",
|
||||
"\n",
|
||||
" dataset_op = gcc_aip.ImageDatasetCreateOp(\n",
|
||||
" project=project,\n",
|
||||
@@ -1081,21 +1128,136 @@
|
||||
" model_display_name=display_name,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" endpoint_op = gcc_aip.EndpointCreateOp(\n",
|
||||
" batch_op = ModelBatchPredictOp(\n",
|
||||
" project=project,\n",
|
||||
" job_display_name=\"batch_predict_job\",\n",
|
||||
" model=training_op.outputs[\"model\"],\n",
|
||||
" gcs_source_uris=batch_files,\n",
|
||||
" gcs_destination_output_uri_prefix=bucket,\n",
|
||||
" instances_format=\"jsonl\",\n",
|
||||
" predictions_format=\"jsonl\",\n",
|
||||
" model_parameters={},\n",
|
||||
" machine_type=DEPLOY_COMPUTE,\n",
|
||||
" starting_replica_count=1,\n",
|
||||
" max_replica_count=1,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" endpoint_op = EndpointCreateOp(\n",
|
||||
" project=project,\n",
|
||||
" location=region,\n",
|
||||
" display_name=display_name,\n",
|
||||
" ).after(training_op)\n",
|
||||
" ).after(batch_op)\n",
|
||||
"\n",
|
||||
" deploy_op = gcc_aip.ModelDeployOp(\n",
|
||||
" deploy_op = ModelDeployOp(\n",
|
||||
" model=training_op.outputs[\"model\"],\n",
|
||||
" endpoint=endpoint_op.outputs[\"endpoint\"],\n",
|
||||
" dedicated_resources_min_replica_count=1,\n",
|
||||
" dedicated_resources_max_replica_count=1,\n",
|
||||
" dedicated_resources_machine_type=\"n1-standard-4\",\n",
|
||||
" dedicated_resources_machine_type=DEPLOY_COMPUTE,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "get_test_items:batch_prediction"
|
||||
},
|
||||
"source": [
|
||||
"### Get test item(s)\n",
|
||||
"\n",
|
||||
"Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "get_test_items:automl,icn,csv"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"test_items = !gsutil cat $IMPORT_FILE | head -n2\n",
|
||||
"if len(str(test_items[0]).split(\",\")) == 3:\n",
|
||||
" _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
"else:\n",
|
||||
" test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n",
|
||||
" test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n",
|
||||
"\n",
|
||||
"print(test_item_1, test_label_1)\n",
|
||||
"print(test_item_2, test_label_2)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "copy_test_items:batch_prediction"
|
||||
},
|
||||
"source": [
|
||||
"### Copy test item(s)\n",
|
||||
"\n",
|
||||
"For the batch prediction, copy the test items over to your Cloud Storage bucket."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "copy_test_items:batch_prediction"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"file_1 = test_item_1.split(\"/\")[-1]\n",
|
||||
"file_2 = test_item_2.split(\"/\")[-1]\n",
|
||||
"\n",
|
||||
"! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n",
|
||||
"! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n",
|
||||
"\n",
|
||||
"test_item_1 = BUCKET_NAME + \"/\" + file_1\n",
|
||||
"test_item_2 = BUCKET_NAME + \"/\" + file_2"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "make_batch_file:automl,image"
|
||||
},
|
||||
"source": [
|
||||
"### Make the batch input file\n",
|
||||
"\n",
|
||||
"Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n",
|
||||
"\n",
|
||||
"- `content`: The Cloud Storage path to the image.\n",
|
||||
"- `mime_type`: The content type. In our example, it is a `jpeg` file.\n",
|
||||
"\n",
|
||||
"For example:\n",
|
||||
"\n",
|
||||
" {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "make_batch_file:automl,image"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"\n",
|
||||
"import tensorflow as tf\n",
|
||||
"\n",
|
||||
"gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n",
|
||||
"with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n",
|
||||
" data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n",
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
" data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n",
|
||||
" f.write(json.dumps(data) + \"\\n\")\n",
|
||||
"\n",
|
||||
"print(gcs_input_uri)\n",
|
||||
"! gsutil cat $gcs_input_uri"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
@@ -1107,6 +1269,7 @@
|
||||
"Next, you compile the pipeline and then exeute it. The pipeline takes the following parameters, which are passed as the dictionary `parameter_values`:\n",
|
||||
"\n",
|
||||
"- `import_file`: The Cloud Storage path to the dataset index file.\n",
|
||||
"- `batch_files`: A list of Cloud Storage paths to the input batch files.\n",
|
||||
"- `display_name`: The display name for the generated Vertex AI resources.\n",
|
||||
"- `python_package`: The Python package for the custom training job.\n",
|
||||
"- `python_module`: The Python module in the package to execute.\n",
|
||||
@@ -1132,6 +1295,7 @@
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" \"import_file\": IMPORT_FILE,\n",
|
||||
" \"batch_files\": [gcs_input_uri],\n",
|
||||
" \"display_name\": \"flowers\" + TIMESTAMP,\n",
|
||||
" \"python_package\": f\"{BUCKET_NAME}/trainer_flowers.tar.gz\",\n",
|
||||
" \"python_module\": \"trainer.task\",\n",
|
||||
@@ -1207,17 +1371,425 @@
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"print(\"imagedataset-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"imagedataset-create\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"image-dataset-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"image-dataset-create\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"custompythonpackagetrainingjob-run\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"custompythonpackagetrainingjob-run\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"endpoint-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"endpoint-create\")\n",
|
||||
"print(\"\\n\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-deploy\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-deploy\")"
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-deploy\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-batch-predict\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-batch-predict\")\n",
|
||||
"output = !gsutil cat $artifacts\n",
|
||||
"output = json.loads(output[0])\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\n",
|
||||
" output[\"artifacts\"][\"batchpredictionjob\"][\"artifacts\"][0][\"metadata\"][\n",
|
||||
" \"gcsOutputDirectory\"\n",
|
||||
" ]\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "delete_pipeline"
|
||||
},
|
||||
"source": [
|
||||
"### Delete a pipeline job\n",
|
||||
"\n",
|
||||
"After a pipeline job is completed, you can delete the pipeline job with the method `delete()`. Prior to completion, a pipeline job can be canceled with the method `cancel()`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "delete_pipeline"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"pipeline.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_custom_training_job_op_from_component:intro"
|
||||
},
|
||||
"source": [
|
||||
"## Using `create_custom_training_job_op_from_component`\n",
|
||||
"\n",
|
||||
"An alternative approach is for you to create your own component to do custom training, instead of creating a Python package and executing it with a `CustomPythonPackageTrainingOp`. In this case, what would have been inside the Python package is instead directly embedded in your component.\n",
|
||||
"\n",
|
||||
"You might do this for example if you are early on in the development of the training package and you want speed and convenience over scaling. One issue with this is that when executed it will only be seen and tracked as a component artifact, versus being seen and tracked as a CustomTrainingJob.\n",
|
||||
"\n",
|
||||
"The utility `google_cloud_pipeline_components.experimental.custom_job.utils.create_custom_training_job_op_from_component` provides you the benefits of both. This utility takes as input your custom training component and outputs a conversion to a `CustomTrainingJobOp` component."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_component:custom_train_model"
|
||||
},
|
||||
"source": [
|
||||
"### Create a custom training job component\n",
|
||||
"\n",
|
||||
"First, you create a custom training component `custom_train_model`. In this component, you will use a very simple script to train a CIFAR-10 model. The script has very few bells and whistles otherthan:\n",
|
||||
"\n",
|
||||
"- Setting the learning rate, number of epochs, batch size and number of steps per epoch as parameters to your component.\n",
|
||||
"- Setting the Cloud Storage location to save the trained model artifacts to.\n",
|
||||
"\n",
|
||||
"Note, you set the default value of `model_dir` to a null string. The reason you do this, is that the training service may alternately specifiy the location with the environment variable `AIP_MODEL_DIR`. The code logic is: if the paraneter `model_dir` is set (non-empty), use that value; otherwise use the location specified by the environment variable `AIP_MODEL_DIR`.\n",
|
||||
"\n",
|
||||
"Once the model is trained, you need to know where the model artifacts are located. Since their location can be either that of the `model_dir` parameter or the environment variable `AIP_MODEL_DIR`. You do this with the component `model_artifacts()`. If the `model_dir` parameter is non-empty, then return its value; otherwise construct the value of AIP_MODEL_DIR setting from the `base_output_directory` parameter."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_component:custom_train_model"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components.v1.custom_job import utils\n",
|
||||
"from kfp.v2.dsl import Artifact\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@component(\n",
|
||||
" base_image=\"tensorflow/tensorflow:latest\",\n",
|
||||
" packages_to_install=[\"tensorflow_datasets\"],\n",
|
||||
")\n",
|
||||
"def custom_train_model(\n",
|
||||
" model_dir: str = \"\",\n",
|
||||
" lr: float = 0.01,\n",
|
||||
" epochs: int = 10,\n",
|
||||
" steps: int = 200,\n",
|
||||
" batch_size: int = 64,\n",
|
||||
"):\n",
|
||||
" import os\n",
|
||||
"\n",
|
||||
" import tensorflow as tf\n",
|
||||
" import tensorflow_datasets as tfds\n",
|
||||
"\n",
|
||||
" if model_dir == \"\":\n",
|
||||
" model_dir = os.getenv(\"AIP_MODEL_DIR\")\n",
|
||||
"\n",
|
||||
" # Preparing dataset\n",
|
||||
" BUFFER_SIZE = 10000\n",
|
||||
"\n",
|
||||
" def get_data():\n",
|
||||
"\n",
|
||||
" # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n",
|
||||
" def scale(image, label):\n",
|
||||
" image = tf.cast(image, tf.float32)\n",
|
||||
" image /= 255.0\n",
|
||||
" return image, label\n",
|
||||
"\n",
|
||||
" datasets, info = tfds.load(name=\"cifar10\", with_info=True, as_supervised=True)\n",
|
||||
" return datasets[\"train\"].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n",
|
||||
"\n",
|
||||
" # Build the Keras model\n",
|
||||
" def get_model():\n",
|
||||
" model = tf.keras.Sequential(\n",
|
||||
" [\n",
|
||||
" tf.keras.layers.Conv2D(\n",
|
||||
" 32, 3, activation=\"relu\", input_shape=(32, 32, 3)\n",
|
||||
" ),\n",
|
||||
" tf.keras.layers.MaxPooling2D(),\n",
|
||||
" tf.keras.layers.Conv2D(32, 3, activation=\"relu\"),\n",
|
||||
" tf.keras.layers.MaxPooling2D(),\n",
|
||||
" tf.keras.layers.Flatten(),\n",
|
||||
" tf.keras.layers.Dense(10, activation=\"softmax\"),\n",
|
||||
" ]\n",
|
||||
" )\n",
|
||||
" model.compile(\n",
|
||||
" loss=tf.keras.losses.sparse_categorical_crossentropy,\n",
|
||||
" optimizer=tf.keras.optimizers.SGD(learning_rate=lr),\n",
|
||||
" metrics=[\"accuracy\"],\n",
|
||||
" )\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
" def train_model(model, train_dataset):\n",
|
||||
" model.fit(x=train_dataset, epochs=epochs, steps_per_epoch=steps)\n",
|
||||
" return model\n",
|
||||
"\n",
|
||||
" train_dataset = get_data().batch(batch_size)\n",
|
||||
"\n",
|
||||
" model = get_model()\n",
|
||||
"\n",
|
||||
" model = train_model(model, train_dataset)\n",
|
||||
"\n",
|
||||
" model.save(model_dir)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@component()\n",
|
||||
"def model_artifacts(model_dir: str, base_output_directory: str) -> str:\n",
|
||||
" # location of model artifacts overridden by model_dir parameter\n",
|
||||
" if model_dir != \"\":\n",
|
||||
" return model_dir\n",
|
||||
" # location of model artifacts specified by base_output_directory\n",
|
||||
" else:\n",
|
||||
" return base_output_directory + \"/model\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_custom_training_job_op_from_component"
|
||||
},
|
||||
"source": [
|
||||
"### Convert your custom training component to a predefined CustomTrainingJobOp component\n",
|
||||
"\n",
|
||||
"Next, use the utility to convert your custom training component to a CustomTrainingJobOp component, as the required parameter. There are some additional optional keyword parameters to overide default settings in the worker pool specification, service account, and optional setting tensorboard instance and encryption key.\n",
|
||||
"\n",
|
||||
"Learn more about [create_custom_training_job_op_from_component reference](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-0.2.0/google_cloud_pipeline_components.experimental.custom_job.html)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_custom_training_job_op_from_component"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"custom_job_training_op = utils.create_custom_training_job_op_from_component(\n",
|
||||
" custom_train_model, replica_count=1\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "create_custom_pipeline:custom_train_op"
|
||||
},
|
||||
"source": [
|
||||
"## Construct custom training pipeline\n",
|
||||
"\n",
|
||||
"In the example below, you construct a pipeline for training a custom model using:\n",
|
||||
"\n",
|
||||
"1. Pipeline arguments, specify the locations of:\n",
|
||||
" - `display_name`: The human readable name for the model and endpoint.\n",
|
||||
" - `model_dir`: Optionally location (override) for saving the model artifacts\n",
|
||||
" - `epochs`: The number of epochs.\n",
|
||||
" - `steps`: The number of steps per epoch.\n",
|
||||
" - `lr`: The learning rate.\n",
|
||||
" - `project`: The project for executing the pipeline components.\n",
|
||||
" - `location`: The location for executing the pipeline components.\n",
|
||||
" - `deploy_image`: The serving container.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"2. Use the converted `custom_job_training_op` to train the custom model.\n",
|
||||
" - If model_dir is non-empty string, it will override the setting of base_output_directory.\n",
|
||||
"\n",
|
||||
"3. Use the custom component `model_artifacts` to determine the location of the model artifacts.\n",
|
||||
" - If model_dir is non-empty string, return its location.\n",
|
||||
" - Otherwise, return the derived location from base_output_directory.\n",
|
||||
"\n",
|
||||
"4. Use the prebuilt component `ModelUploadOp` to create a `Vertex AI Model` resource from the model artifacts.\n",
|
||||
" - The location of the model artifacts is the output from the component `model_artifacts`.\n",
|
||||
" - The `after(custom_job_op)` is added to serialize its execution, so its only executed if the training operation completes successfully.\n",
|
||||
"\n",
|
||||
"5. Use the prebuilt component `EndpointCreateOp` to create a `Vertex AI Endpoint` to deploy the trained model to, where:\n",
|
||||
" - Since the component has no dependencies on other components, by default it would be executed in parallel with the model training.\n",
|
||||
" - The `after(model_op)` is added to serialize its execution, so its only executed if the training operation completes successfully.\n",
|
||||
" - The component returns the endpoint resource as `outputs[\"endpoint\"]`.\n",
|
||||
"\n",
|
||||
"6. Use the prebuilt component `ModelDeployOp` to deploy the trained `Vertex AI Model` to, where:\n",
|
||||
" - The display name for the dataset is passed into the pipeline.\n",
|
||||
" - The model is the output from the `ModelUploadOp`.\n",
|
||||
" - The endpoint is the output from the `EndpointCreateOp`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "create_custom_pipeline:custom_train_op"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components import aiplatform as gcc_aip\n",
|
||||
"\n",
|
||||
"PIPELINE_ROOT = \"{}/pipeline_root/custom_cifar10_training\".format(BUCKET_NAME)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"@dsl.pipeline(name=\"custom-model-training-sample-pipeline\")\n",
|
||||
"def pipeline(\n",
|
||||
" display_name: str,\n",
|
||||
" model_dir: str = \"\",\n",
|
||||
" lr: float = 0.01,\n",
|
||||
" epochs: int = 10,\n",
|
||||
" steps: int = 200,\n",
|
||||
" project: str = PROJECT_ID,\n",
|
||||
" location: str = REGION,\n",
|
||||
" deploy_image: str = \"us-docker.pkg.dev/cloud-aiplatform/prediction/tf2-cpu.2-3:latest\",\n",
|
||||
"):\n",
|
||||
" custom_job_op = custom_job_training_op(\n",
|
||||
" model_dir=model_dir,\n",
|
||||
" lr=lr,\n",
|
||||
" epochs=epochs,\n",
|
||||
" steps=steps,\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" base_output_directory=PIPELINE_ROOT,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" artifacts_op = model_artifacts(model_dir, PIPELINE_ROOT)\n",
|
||||
"\n",
|
||||
" model_upload_op = gcc_aip.ModelUploadOp(\n",
|
||||
" project=project,\n",
|
||||
" display_name=display_name,\n",
|
||||
" artifact_uri=artifacts_op.output,\n",
|
||||
" serving_container_image_uri=deploy_image,\n",
|
||||
" ).after(custom_job_op)\n",
|
||||
"\n",
|
||||
" endpoint_op = gcc_aip.EndpointCreateOp(\n",
|
||||
" project=project,\n",
|
||||
" location=location,\n",
|
||||
" display_name=display_name,\n",
|
||||
" ).after(model_upload_op)\n",
|
||||
"\n",
|
||||
" deploy_op = gcc_aip.ModelDeployOp(\n",
|
||||
" model=model_upload_op.outputs[\"model\"],\n",
|
||||
" endpoint=endpoint_op.outputs[\"endpoint\"],\n",
|
||||
" dedicated_resources_min_replica_count=1,\n",
|
||||
" dedicated_resources_max_replica_count=1,\n",
|
||||
" dedicated_resources_machine_type=DEPLOY_COMPUTE,\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "run_custom_train_pipeline:custom_train_job"
|
||||
},
|
||||
"source": [
|
||||
"### Compile and execute the pipeline\n",
|
||||
"\n",
|
||||
"Next, you compile the pipeline and then exeute it. The pipeline takes the following parameters, which are passed as the dictionary `parameter_values`:\n",
|
||||
"\n",
|
||||
"- `display_name`: The display name for the generated Vertex AI resources.\n",
|
||||
"- `epochs`: The number of epochs.\n",
|
||||
"- `project`: The project ID.\n",
|
||||
"- `region`: The region.\n",
|
||||
"\n",
|
||||
"*Note:* In this execution, you do not override the location of the model artifacts -- i.e., model_dir parameter is a null string."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "run_custom_train_pipeline:custom_train_job"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"compiler.Compiler().compile(\n",
|
||||
" pipeline_func=pipeline, package_path=\"custom_cifar10_training.json\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"pipeline = aip.PipelineJob(\n",
|
||||
" display_name=\"cifar10-custom_training\",\n",
|
||||
" template_path=\"custom_cifar10_training.json\",\n",
|
||||
" pipeline_root=PIPELINE_ROOT,\n",
|
||||
" parameter_values={\n",
|
||||
" \"display_name\": \"simple-example\",\n",
|
||||
" \"epochs\": 20,\n",
|
||||
" \"project\": PROJECT_ID,\n",
|
||||
" \"location\": REGION,\n",
|
||||
" },\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"pipeline.run()\n",
|
||||
"\n",
|
||||
"! rm -f custom_cifar10_training.json"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "view_pipeline_results:custom_train_job"
|
||||
},
|
||||
"source": [
|
||||
"### View custom model training pipeline results\n",
|
||||
"\n",
|
||||
"Finally, you will view the artifact outputs of each task in the pipeline."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "view_pipeline_results:custom_train_job"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"PROJECT_NUMBER = pipeline.gca_resource.name.split(\"/\")[1]\n",
|
||||
"print(PROJECT_NUMBER)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def print_pipeline_output(job, output_task_name):\n",
|
||||
" JOB_ID = job.name\n",
|
||||
" print(JOB_ID)\n",
|
||||
" for _ in range(len(job.gca_resource.job_detail.task_details)):\n",
|
||||
" TASK_ID = job.gca_resource.job_detail.task_details[_].task_id\n",
|
||||
" EXECUTE_OUTPUT = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/executor_output.json\"\n",
|
||||
" )\n",
|
||||
" GCP_RESOURCES = (\n",
|
||||
" PIPELINE_ROOT\n",
|
||||
" + \"/\"\n",
|
||||
" + PROJECT_NUMBER\n",
|
||||
" + \"/\"\n",
|
||||
" + JOB_ID\n",
|
||||
" + \"/\"\n",
|
||||
" + output_task_name\n",
|
||||
" + \"_\"\n",
|
||||
" + str(TASK_ID)\n",
|
||||
" + \"/gcp_resources\"\n",
|
||||
" )\n",
|
||||
" if tf.io.gfile.exists(EXECUTE_OUTPUT):\n",
|
||||
" ! gsutil cat $EXECUTE_OUTPUT\n",
|
||||
" break\n",
|
||||
" elif tf.io.gfile.exists(GCP_RESOURCES):\n",
|
||||
" ! gsutil cat $GCP_RESOURCES\n",
|
||||
" break\n",
|
||||
"\n",
|
||||
" return EXECUTE_OUTPUT\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"print(\"custom-train-model\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"custom-train-model\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-artifacts\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-artifacts\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-upload\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-upload\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"endpoint-create\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"endpoint-create\")\n",
|
||||
"print(\"\\n\\n\")\n",
|
||||
"print(\"model-deploy\")\n",
|
||||
"artifacts = print_pipeline_output(pipeline, \"model-deploy\")\n",
|
||||
"print(\"\\n\\n\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -1294,6 +1866,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2021 Google LLC\n",
|
||||
"# Copyright 2022 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,13 +33,13 @@
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_dataflow_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_dataflow_pipeline_components.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/tree/master/notebooks/official/automl/ml_ops_stage3/get_started_with_dataflow_pipeline_components.ipynb\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/ai/platform/notebooks/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage3/get_started_with_dataflow_pipeline_components.ipynb\">\n",
|
||||
" Open in Google Cloud Notebooks\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -119,12 +119,17 @@
|
||||
" ! pip3 install -U tensorflow-transform==1.2 $USER_FLAG\n",
|
||||
" ! pip3 install -U tensorflow-io==0.18 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform[tensorboard] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-pipeline-components $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-bigquery $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade google-cloud-logging $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade apache-beam[gcp] $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade pyarrow $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade cloudml-hypertune $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG"
|
||||
" ! pip3 install --upgrade kfp $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade torchvision $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade rpy2 $USER_FLAG\n",
|
||||
" ! pip3 install --upgrade python-tabulate $USER_FLAG\n",
|
||||
" ! pip3 install -U opencv-python-headless==4.5.2.52 $USER_FLAG"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -221,7 +226,7 @@
|
||||
"\n",
|
||||
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
|
||||
"\n",
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
|
||||
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -374,7 +379,7 @@
|
||||
"):\n",
|
||||
" # Get your GCP project id from gcloud\n",
|
||||
" shell_output = !gcloud auth list 2>/dev/null\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].strip()\n",
|
||||
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
|
||||
" print(\"Service Account:\", SERVICE_ACCOUNT)"
|
||||
]
|
||||
},
|
||||
@@ -448,9 +453,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google_cloud_pipeline_components.experimental.dataflow import \\\n",
|
||||
" DataflowPythonJobOp\n",
|
||||
"from google_cloud_pipeline_components.experimental.wait_gcp_resources import \\\n",
|
||||
"from google_cloud_pipeline_components.v1.dataflow import DataflowPythonJobOp\n",
|
||||
"from google_cloud_pipeline_components.v1.wait_gcp_resources import \\\n",
|
||||
" WaitGcpResourcesOp"
|
||||
]
|
||||
},
|
||||
@@ -644,7 +648,8 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%%writefile requirements.txt\n",
|
||||
"apache-beam"
|
||||
"apache-beam\n",
|
||||
"future"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -907,7 +912,8 @@
|
||||
"source": [
|
||||
"%%writefile requirements.txt\n",
|
||||
"apache-beam\n",
|
||||
"tensorflow-transform==1.2.0"
|
||||
"tensorflow-transform==1.2.0\n",
|
||||
"future"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -934,6 +940,7 @@
|
||||
"\n",
|
||||
"REQUIRED_PACKAGES = [\n",
|
||||
" 'tensorflow-transform==1.2.0',\n",
|
||||
" 'future'\n",
|
||||
"]\n",
|
||||
"PACKAGE_NAME = 'my_package'\n",
|
||||
"PACKAGE_VERSION = '0.0.1'\n",
|
||||
@@ -1050,9 +1057,7 @@
|
||||
" ],\n",
|
||||
" requirements_file_path: str = GCS_REQUIREMENTS_TXT,\n",
|
||||
"):\n",
|
||||
" DataflowPythonJobOp.component_spec.implementation.container.image = (\n",
|
||||
" \"gcr.io/ml-pipeline/google-cloud-pipeline-components:v0.2.0_dataflow_logs_fix\"\n",
|
||||
" )\n",
|
||||
" # DataflowPythonJobOp.component_spec.implementation.container.image = \"gcr.io/ml-pipeline/google-cloud-pipeline-components:v0.2.0_dataflow_logs_fix\"\n",
|
||||
" dataflow_python_op = DataflowPythonJobOp(\n",
|
||||
" project=project_id,\n",
|
||||
" location=location,\n",
|
||||
@@ -1157,6 +1162,7 @@
|
||||
" # Delete the endpoint using the Vertex endpoint object\n",
|
||||
" try:\n",
|
||||
" if \"endpoint\" in globals():\n",
|
||||
" endpoint.undeploy_all()\n",
|
||||
" endpoint.delete()\n",
|
||||
" except Exception as e:\n",
|
||||
" print(e)\n",
|
||||
|
||||