Compare commits

..
73 changed files with 8715 additions and 23921 deletions
@@ -38,7 +38,7 @@ from utils import NotebookProcessors, util
# A buffer so that workers finish before the orchestrating job
WORKER_TIMEOUT_BUFFER_IN_SECONDS: int = 60 * 60
PYTHON_VERSION = "3.9" # Set default python version
PYTHON_VERSION = "3.9" # Set default python version
def format_timedelta(delta: datetime.timedelta) -> str:
@@ -102,7 +102,6 @@ def _process_notebook(
"VPC_NETWORK": variable_vpc_network,
},
)
unique_strings_preprocessor = NotebookProcessors.UniqueStringsPreprocessor()
# Use no-execute preprocessor
(
@@ -111,7 +110,6 @@ def _process_notebook(
) = remove_no_execute_cells_preprocessor.preprocess(nb)
(nb, resources) = update_variables_preprocessor.preprocess(nb, resources)
(nb, resources) = unique_strings_preprocessor.preprocess(nb, resources)
with open(notebook_path, mode="w", encoding="utf-8") as new_file:
nbformat.write(nb, new_file)
@@ -129,15 +127,13 @@ def _get_notebook_python_version(notebook_path: str) -> str:
src = file.read()
nb_json = json.loads(src)
# Iterate over the cells in the ipynb
for cell in nb_json["cells"]:
if cell["cell_type"] == "markdown":
markdown = str.join("", cell["source"])
#Iterate over the cells in the ipynb
for cell in nb_json['cells']:
if cell['cell_type'] == 'markdown':
markdown = str.join('', cell['source'])
# Look for the python version specification pattern
re_match = re.search(
"python version = (\d\.\d)", markdown, flags=re.IGNORECASE
)
re_match = re.search('python version = (\d\.\d)', markdown, flags=re.IGNORECASE)
if re_match:
# get the version number
python_version = re_match.group(1)
@@ -205,9 +201,7 @@ def process_and_execute_notebook(
operation = None
try:
# Get the python version for running the notebook if specified
notebook_exec_python_version = _get_notebook_python_version(
notebook_path=notebook
)
notebook_exec_python_version = _get_notebook_python_version(notebook_path=notebook)
print(f"Running notebook with python {notebook_exec_python_version}")
# Pre-process notebook by substituting variable names
@@ -236,7 +230,7 @@ def process_and_execute_notebook(
private_pool_id=private_pool_id,
private_pool_region=variable_region,
timeout_in_seconds=timeout_in_seconds,
python_version=notebook_exec_python_version,
python_version=notebook_exec_python_version
)
operation_metadata = BuildOperationMetadata(mapping=operation.metadata)
@@ -449,7 +443,7 @@ def process_and_execute_notebooks(
result.log_url,
result.output_uri,
result.output_uri_web,
result.logs_bucket,
result.logs_bucket
]
for result in results_sorted
],
@@ -460,34 +454,34 @@ def process_and_execute_notebooks(
"log_url",
"output_uri",
"output_uri_web",
"logs_bucket",
"logs_bucket"
],
)
)
if len(notebooks) == 1:
print("=" * 100)
print("The notebook execution build log:\n")
print("=" * 100)
print("="*100)
print("The notebook execution build log:\n")
print("="*100)
build_id = results_sorted[0].build_id
logs_bucket_name = (results_sorted[0].logs_bucket).removeprefix("gs://")
log_file_name = f"log-{build_id}.txt"
build_id = results_sorted[0].build_id
logs_bucket_name = (results_sorted[0].logs_bucket).removeprefix("gs://")
log_file_name = f"log-{build_id}.txt"
log_contents = util.download_blob_into_memory(
bucket_name=logs_bucket_name,
blob_name=log_file_name,
download_as_text=True,
log_contents = util.download_blob_into_memory(
bucket_name=logs_bucket_name,
blob_name=log_file_name,
download_as_text=True
)
# Remove extra steps from the log
match = re.search("starting Step #4", log_contents, flags=re.IGNORECASE)
# Remove extra steps from the log
match = re.search("starting Step #4", log_contents, flags=re.IGNORECASE)
if match is not None:
match_index = match.span()[0]
print(log_contents[match_index:])
else:
print(log_contents)
if match is not None:
match_index = match.span()[0]
print(log_contents[match_index:])
else:
print(log_contents)
print("\n=== END RESULTS===\n")
+1
View File
@@ -2,4 +2,5 @@ notebooks/official/vizier/gapic-vizier-multi-objective-optimization.ipynb
notebooks/official/pipelines/lightweight_functions_component_io_kfp.ipynb
notebooks/official/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb
notebooks/official/custom/custom-tabular-bq-managed-dataset.ipynb
notebooks/official/matching_engine/sdk_matching_engine_for_indexing.ipynb
.cloud-build/tests/python_version_test.ipynb
-35
View File
@@ -14,8 +14,6 @@
# limitations under the License.
from typing import Dict
import random
import string
from nbconvert.preprocessors import Preprocessor
@@ -65,36 +63,3 @@ class UpdateVariablesPreprocessor(Preprocessor):
executable_cells.append(cell)
notebook.cells = executable_cells
return notebook, resources
# Generate a uuid of a specifed length
def generate_uuid(length: int = 8) -> str:
return "".join(random.choices(string.ascii_lowercase + string.digits, k=length))
class UniqueStringsPreprocessor(Preprocessor):
# A preprocessor that replaces strings that end with "-unique" or "_unique" with a uuid.
@staticmethod
def update_unique_strings(content: str):
# Replace strings that end with "-unique" or "_unique" with a uuid.
unique_id = generate_uuid()
return (
content.replace('-unique"', f'-{unique_id}"')
.replace("-unique'", f'-{unique_id}"')
.replace('_unique"', f'_{unique_id}"')
.replace("_unique'", f'_{unique_id}"')
)
def preprocess(self, notebook, resources=None):
executable_cells = []
for cell in notebook.cells:
if cell.cell_type == "code":
cell.source = self.update_unique_strings(
content=cell.source,
)
executable_cells.append(cell)
notebook.cells = executable_cells
return notebook, resources
@@ -40,3 +40,65 @@ def get_updated_value(content: str, variable_name: str, variable_value: str) ->
content,
flags=re.M,
)
def test_update_value():
new_content = get_updated_value(
content='asdf\nPROJECT_ID = "[your-project-id]" #@param {type:"string"} \nasdf',
variable_name="PROJECT_ID",
variable_value="sample-project",
)
assert (
new_content
== 'asdf\nPROJECT_ID = "sample-project" #@param {type:"string"} \nasdf'
)
def test_update_value_single_quotes():
new_content = get_updated_value(
content="PROJECT_ID = '[your-project-id]'",
variable_name="PROJECT_ID",
variable_value="sample-project",
)
assert new_content == "PROJECT_ID = 'sample-project'"
def test_update_value_avoidance():
new_content = get_updated_value(
content="PROJECT_ID = shell_output[0] ",
variable_name="PROJECT_ID",
variable_value="sample-project",
)
assert new_content == "PROJECT_ID = shell_output[0] "
def test_region():
new_content = get_updated_value(
content='REGION = "[your-region]" # @param {type:"string"}',
variable_name="REGION",
variable_value="us-central1",
)
assert new_content == 'REGION = "us-central1" # @param {type:"string"}'
def test_region_equal_equals_ignore():
# Tests that == is ignored
new_content = get_updated_value(
content='REGION == "[your-region]" # @param {type:"string"}',
variable_name="REGION",
variable_value="us-central1",
)
assert new_content == 'REGION == "[your-region]" # @param {type:"string"}'
def test_service_account():
# Tests that == is ignored
new_content = get_updated_value(
content='SERVICE_ACCOUNT = "[your-service-account]" # @param {type:"string"}',
variable_name="SERVICE_ACCOUNT",
variable_value="12345-compute@developer.gserviceaccount.com",
)
assert (
new_content
== 'SERVICE_ACCOUNT = "12345-compute@developer.gserviceaccount.com" # @param {type:"string"}'
)
@@ -1,14 +0,0 @@
from utils import NotebookProcessors
def test_update_value():
# Test that the content was updated
preprocessor = NotebookProcessors.UniqueStringsPreprocessor()
content = 'PROJECT_ID = "your-project-id-unique"'
new_content = preprocessor.update_unique_strings(content)
assert new_content != content
assert new_content.startswith('PROJECT_ID = "your-project-id-')
assert new_content.endswith('"')
@@ -1,63 +0,0 @@
from utils import UpdateNotebookVariables
def test_update_value():
new_content = UpdateNotebookVariables.get_updated_value(
content='asdf\nPROJECT_ID = "[your-project-id]" #@param {type:"string"} \nasdf',
variable_name="PROJECT_ID",
variable_value="sample-project",
)
assert (
new_content
== 'asdf\nPROJECT_ID = "sample-project" #@param {type:"string"} \nasdf'
)
def test_update_value_single_quotes():
new_content = UpdateNotebookVariables.get_updated_value(
content="PROJECT_ID = '[your-project-id]'",
variable_name="PROJECT_ID",
variable_value="sample-project",
)
assert new_content == "PROJECT_ID = 'sample-project'"
def test_update_value_avoidance():
new_content = UpdateNotebookVariables.get_updated_value(
content="PROJECT_ID = shell_output[0] ",
variable_name="PROJECT_ID",
variable_value="sample-project",
)
assert new_content == "PROJECT_ID = shell_output[0] "
def test_region():
new_content = UpdateNotebookVariables.get_updated_value(
content='REGION = "[your-region]" # @param {type:"string"}',
variable_name="REGION",
variable_value="us-central1",
)
assert new_content == 'REGION = "us-central1" # @param {type:"string"}'
def test_region_equal_equals_ignore():
# Tests that == is ignored
new_content = UpdateNotebookVariables.get_updated_value(
content='REGION == "[your-region]" # @param {type:"string"}',
variable_name="REGION",
variable_value="us-central1",
)
assert new_content == 'REGION == "[your-region]" # @param {type:"string"}'
def test_service_account():
# Tests that == is ignored
new_content = UpdateNotebookVariables.get_updated_value(
content='SERVICE_ACCOUNT = "[your-service-account]" # @param {type:"string"}',
variable_name="SERVICE_ACCOUNT",
variable_value="12345-compute@developer.gserviceaccount.com",
)
assert (
new_content
== 'SERVICE_ACCOUNT = "12345-compute@developer.gserviceaccount.com" # @param {type:"string"}'
)
+9 -4
View File
@@ -61,7 +61,9 @@ def archive_code_and_upload(staging_bucket: str):
def download_blob_into_memory(
bucket_name: str, blob_name: str, download_as_text: Optional[bool] = False
bucket_name: str,
blob_name: str,
download_as_text: Optional[bool]=False
) -> Union[bytes, str]:
"""
Downloads a blob into memory as byte or as text if
@@ -77,10 +79,13 @@ def download_blob_into_memory(
# Download the blob content
if download_as_text:
contents = blob.download_as_text()
contents = blob.download_as_text()
else:
contents = blob.download_as_bytes()
contents = blob.download_as_bytes()
print(f"Downloaded storage object {blob_name} from bucket {bucket_name}.")
print(
f"Downloaded storage object {blob_name} from bucket {bucket_name}."
)
return contents
+3 -3
View File
@@ -2,9 +2,9 @@ git+https://github.com/tensorflow/docs
ipython
jupyter
nbconvert
black==22.10.0
pyupgrade==2.38.4
black==22.6.0
pyupgrade==2.34.0
isort==5.10.1
flake8==4.0.1
nbqa==1.5.3
nbqa==1.4.0
-1
View File
@@ -6,4 +6,3 @@
/sklearn_text_classification_from_script_using_vertex_sdk @maxhardt
/pluto_on_workbench @wkharold
/cpr-examples @samthrasher
/Train_tabular_models_with_many_frameworks_and_import_to_Vertex_AI_using_Pipelines @Ark-kun
@@ -1,83 +0,0 @@
name: Train tabular classification logistic regression model using Scikit learn pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_classification_logistic_regression_model_using_Scikit_learn_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":40,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":40,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":40,"y":250,"width":180,"height":54}'
Binarize column using Pandas on CSV data:
componentRef:
digest: d699afd4d7cae862708717cc160f4394ed0c04e536e9515923ef1e8865f01d44
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
column_name: tips
predicate: '> 0'
new_column_name: class
annotations:
editor.position: '{"x":40,"y":380,"width":180,"height":54}'
Train logistic regression model using scikit learn from CSV:
componentRef:
digest: a864625a822e4b1c8ef6fe4ae1454fd90f15438f70a6712bb4c30e0dda4d35b7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/cb44b75c9c062fcc40c2b905b2024b4493dbc62b/components/ML_frameworks/Scikit_learn/Train_logistic_regression_model/from_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: transformed_table
taskId: Binarize column using Pandas on CSV data
type: CSV
label_column_name: class
annotations:
editor.position: '{"x":40,"y":510,"width":180,"height":70}'
Upload Scikit learn pickle model to Google Cloud Vertex AI:
componentRef:
digest: 81c91c8d7d21ec97e0872f669d68bd89edea87279d703685db54aa94743bebcd
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train logistic regression model using scikit learn from CSV
type: ScikitLearnPickleModel
annotations:
editor.position: '{"x":40,"y":660,"width":180,"height":70}'
outputValues: {}
@@ -1,73 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
binarize_column_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml")
train_logistic_regression_model_using_scikit_learn_from_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/cb44b75c9c062fcc40c2b905b2024b4493dbc62b/components/ML_frameworks/Scikit_learn/Train_logistic_regression_model/from_CSV/component.yaml")
upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_classification_logistic_regression_model_using_Scikit_learn_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
# Deploying the model might incur additional costs over time
deploy_model = False
classification_label_column = "class"
all_columns = [label_column] + feature_columns
training_data = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
training_data = select_columns_using_Pandas_on_CSV_data_op(
table=training_data,
column_names=all_columns,
).outputs["transformed_table"]
# Cleaning the NaN values.
training_data = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=training_data,
replacement_value="0",
#replacement_type_name="float",
).outputs["transformed_table"]
classification_training_data = binarize_column_using_Pandas_on_CSV_data_op(
table=training_data,
column_name=label_column,
predicate="> 0",
new_column_name=classification_label_column,
).outputs["transformed_table"]
model = train_logistic_regression_model_using_scikit_learn_from_CSV_op(
dataset=classification_training_data,
label_column_name=classification_label_column,
# Optional:
#penalty="l2",
#solver="lbfgs",
#max_iterations=100,
#multi_class_mode="auto",
#random_seed=0,
).outputs["model"]
vertex_model_name = upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op(
model=model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
sklearn_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func = train_tabular_classification_logistic_regression_model_using_Scikit_learn_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,114 +0,0 @@
name: Train tabular classification model using PyTorch pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_classification_model_using_PyTorch_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":240,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":240,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":240,"y":250,"width":180,"height":54}'
Create fully connected pytorch network:
componentRef:
digest: d03d8248fd358a0275ec33568ee7dd7dce576cc112b09dfafe2651e4d97e04a9
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
output_activation_name: sigmoid
annotations:
editor.position: '{"x":40,"y":360,"width":180,"height":54}'
Binarize column using Pandas on CSV data:
componentRef:
digest: d699afd4d7cae862708717cc160f4394ed0c04e536e9515923ef1e8865f01d44
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
column_name: tips
predicate: ' > 0'
new_column_name: class
annotations:
editor.position: '{"x":240,"y":360,"width":180,"height":54}'
Train pytorch model from csv:
componentRef:
digest: 40f3185eb61e9727f41a4e0c05dd3d3b44bd802aa0f378cfc31756560033949a
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Create fully connected pytorch network
type: PyTorchScriptModule
training_data:
taskOutput:
outputName: transformed_table
taskId: Binarize column using Pandas on CSV data
type: CSV
label_column_name: class
loss_function_name: binary_cross_entropy
annotations:
editor.position: '{"x":240,"y":490,"width":180,"height":40}'
Create PyTorch Model Archive with base handler:
componentRef:
digest: 8298b5ee1b0f0879f893add4cf352c8dec7cf9e21bb9db134c91a2d046cdb0ec
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml
arguments:
Model:
taskOutput:
outputName: trained_model
taskId: Train pytorch model from csv
type: PyTorchScriptModule
Model name: model
Model version: '1.0'
annotations:
editor.position: '{"x":240,"y":590,"width":180,"height":54}'
Upload PyTorch model archive to Google Cloud Vertex AI:
componentRef:
digest: 4450212fae7b9001482aca7eb78b28413c205506eccf08a04e7754a8dfa99004
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model_archive:
taskOutput:
outputName: Model archive
taskId: Create PyTorch Model Archive with base handler
type: PyTorchModelArchive
annotations:
editor.position: '{"x":240,"y":720,"width":180,"height":70}'
outputValues: {}
@@ -1,95 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
binarize_column_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml")
create_fully_connected_pytorch_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml")
train_pytorch_model_from_csv_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml")
create_pytorch_model_archive_with_base_handler_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml")
upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_classification_model_using_PyTorch_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
# Deploying the model might incur additional costs over time
deploy_model = False
classification_label_column = "class"
all_columns = [label_column] + feature_columns
training_data = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
training_data = select_columns_using_Pandas_on_CSV_data_op(
table=training_data,
column_names=all_columns,
).outputs["transformed_table"]
# Cleaning the NaN values.
training_data = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=training_data,
replacement_value="0",
#replacement_type_name="float",
).outputs["transformed_table"]
classification_training_data = binarize_column_using_Pandas_on_CSV_data_op(
table=training_data,
column_name=label_column,
predicate=" > 0",
new_column_name=classification_label_column,
).outputs["transformed_table"]
network = create_fully_connected_pytorch_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
output_activation_name="sigmoid",
# output_size=1,
).outputs["model"]
model = train_pytorch_model_from_csv_op(
model=network,
training_data=classification_training_data,
label_column_name=classification_label_column,
loss_function_name="binary_cross_entropy",
# Optional:
#number_of_epochs=1,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
#batch_log_interval=100,
#random_seed=0,
).outputs["trained_model"]
model_archive = create_pytorch_model_archive_with_base_handler_op(
model=model,
# Optional:
# model_name="model",
# model_version="1.0",
).outputs["Model archive"]
vertex_model_name = upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op(
model_archive=model_archive,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func=train_tabular_classification_model_using_PyTorch_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,132 +0,0 @@
name: Train tabular classification model using TensorFlow pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_classification_model_using_TensorFlow_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":40,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":40,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":40,"y":250,"width":180,"height":54}'
Binarize column using Pandas on CSV data:
componentRef:
digest: d699afd4d7cae862708717cc160f4394ed0c04e536e9515923ef1e8865f01d44
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
column_name: tips
predicate: ' > 0'
new_column_name: class
annotations:
editor.position: '{"x":40,"y":370,"width":180,"height":54}'
Split rows into subsets:
componentRef:
digest: a609c3c9196484290f24a1174955f95b27f07a7b458aa5cb8cde28866cb2cb46
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Binarize column using Pandas on CSV data
type: CSV
fraction_1: '0.8'
annotations:
editor.position: '{"x":170,"y":500,"width":180,"height":40}'
Create fully connected tensorflow network:
componentRef:
digest: bfcafbc5ce711b1f69cabf1338212d10d50136a73db9f9f7c984de7b80b4bfb0
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
output_activation_name: sigmoid
annotations:
editor.position: '{"x":370,"y":500,"width":180,"height":54}'
Train model using Keras on CSV:
componentRef:
digest: 42ae60c889034dbad74815653e95b4f7d576b5f47f803173e8679c7b54984609
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Create fully connected tensorflow network
type: TensorflowSavedModel
label_column_name: class
loss_function_name: binary_crossentropy
number_of_epochs: '10'
annotations:
editor.position: '{"x":40,"y":620,"width":180,"height":54}'
Upload Tensorflow model to Google Cloud Vertex AI:
componentRef:
digest: 2e45263ff640b1a688e359b6936e27a81b2407749a84f340af2aa5547e0cb92c
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
annotations:
editor.position: '{"x":40,"y":750,"width":180,"height":54}'
Predict with TensorFlow model on CSV data:
componentRef:
digest: 921bb1563e93a78233b8acceab87055b9154ccf5595d056028cf0396ca224cd4
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
label_column_name: class
annotations:
editor.position: '{"x":240,"y":750,"width":180,"height":54}'
outputValues: {}
@@ -1,106 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
binarize_column_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml")
split_rows_into_subsets_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml")
create_fully_connected_tensorflow_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml")
train_model_using_Keras_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml")
predict_with_TensorFlow_model_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml")
upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_classification_model_using_TensorFlow_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
training_set_fraction = 0.8
# Deploying the model might incur additional costs over time
deploy_model = False
classification_label_column = "class"
all_columns = [label_column] + feature_columns
dataset = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
dataset = select_columns_using_Pandas_on_CSV_data_op(
table=dataset,
column_names=all_columns,
).outputs["transformed_table"]
dataset = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=dataset,
replacement_value="0",
# # Optional:
# column_names=None, # =[...]
).outputs["transformed_table"]
classification_dataset = binarize_column_using_Pandas_on_CSV_data_op(
table=dataset,
column_name=label_column,
predicate=" > 0",
new_column_name=classification_label_column,
).outputs["transformed_table"]
split_task = split_rows_into_subsets_op(
table=classification_dataset,
fraction_1=training_set_fraction,
)
classification_training_data = split_task.outputs["split_1"]
classification_testing_data = split_task.outputs["split_2"]
network = create_fully_connected_tensorflow_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
output_activation_name="sigmoid",
# output_size=1,
).outputs["model"]
model = train_model_using_Keras_on_CSV_op(
training_data=classification_training_data,
model=network,
label_column_name=classification_label_column,
# Optional:
loss_function_name="binary_crossentropy",
number_of_epochs=10,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
#metric_names=["mean_absolute_error"],
#random_seed=0,
).outputs["trained_model"]
predictions = predict_with_TensorFlow_model_on_CSV_data_op(
dataset=classification_testing_data,
model=model,
# label_column_name needs to be set when doing prediction on a dataset that has labels
label_column_name=classification_label_column,
# Optional:
# batch_size=1000,
).outputs["predictions"]
vertex_model_name = upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op(
model=model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func = train_tabular_classification_model_using_TensorFlow_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,115 +0,0 @@
name: Train tabular classification model using XGBoost pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_classification_model_using_XGBoost_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":40,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":40,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":40,"y":250,"width":180,"height":54}'
Binarize column using Pandas on CSV data:
componentRef:
digest: d699afd4d7cae862708717cc160f4394ed0c04e536e9515923ef1e8865f01d44
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
column_name: tips
predicate: '> 0'
new_column_name: class
annotations:
editor.position: '{"x":40,"y":380,"width":180,"height":54}'
Split rows into subsets:
componentRef:
digest: a609c3c9196484290f24a1174955f95b27f07a7b458aa5cb8cde28866cb2cb46
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Binarize column using Pandas on CSV data
type: CSV
fraction_1: '0.8'
annotations:
editor.position: '{"x":170,"y":510,"width":180,"height":40}'
Train XGBoost model on CSV:
componentRef:
digest: 538c5a01eb38deaf532d619f0bbeaff4efc550fe1f0f776fc06791097b68ceac
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: class
objective: binary:logistic
annotations:
editor.position: '{"x":40,"y":630,"width":180,"height":40}'
Upload XGBoost model to Google Cloud Vertex AI:
componentRef:
digest: 5a5a273c403670743820986c03a4175b7cb4595a556524fefcce403656286977
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
annotations:
editor.position: '{"x":40,"y":750,"width":180,"height":54}'
Xgboost predict on CSV:
componentRef:
digest: 0876233a0c7306fefec188bd70f059b46d1fb5aa57be231799570e3bbbdd0d95
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml
arguments:
data:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
label_column_name: class
annotations:
editor.position: '{"x":240,"y":750,"width":180,"height":40}'
outputValues: {}
@@ -1,94 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
binarize_column_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml")
split_rows_into_subsets_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml")
train_XGBoost_model_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml")
xgboost_predict_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml")
upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_classification_model_using_XGBoost_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
training_set_fraction = 0.8
# Deploying the model might incur additional costs over time
deploy_model = False
classification_label_column = "class"
all_columns = [label_column] + feature_columns
dataset = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
dataset = select_columns_using_Pandas_on_CSV_data_op(
table=dataset,
column_names=all_columns,
).outputs["transformed_table"]
dataset = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=dataset,
replacement_value="0",
# # Optional:
# column_names=None, # =[...]
).outputs["transformed_table"]
classification_dataset = binarize_column_using_Pandas_on_CSV_data_op(
table=dataset,
column_name=label_column,
predicate="> 0",
new_column_name=classification_label_column,
).outputs["transformed_table"]
split_task = split_rows_into_subsets_op(
table=classification_dataset,
fraction_1=training_set_fraction,
)
classification_training_data = split_task.outputs["split_1"]
classification_testing_data = split_task.outputs["split_2"]
model = train_XGBoost_model_on_CSV_op(
training_data=classification_training_data,
label_column_name=classification_label_column,
objective="binary:logistic",
# Optional:
#starting_model=None,
#num_iterations=10,
#booster_params={},
#booster="gbtree",
#learning_rate=0.3,
#min_split_loss=0,
#max_depth=6,
).outputs["model"]
# Predicting on the testing data
predictions = xgboost_predict_on_CSV_op(
data=classification_testing_data,
model=model,
# label_column needs to be set when doing prediction on a dataset that has labels
label_column_name=classification_label_column,
).outputs["predictions"]
vertex_model_name = upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op(
model=model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func = train_tabular_classification_model_using_XGBoost_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,257 +0,0 @@
name: Train tabular classification model using all frameworks pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_classification_model_using_all_frameworks_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":550,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":550,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":550,"y":250,"width":180,"height":54}'
Binarize column using Pandas on CSV data:
componentRef:
digest: d699afd4d7cae862708717cc160f4394ed0c04e536e9515923ef1e8865f01d44
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
column_name: tips
predicate: ' > 0'
new_column_name: class
annotations:
editor.position: '{"x":550,"y":380,"width":180,"height":54}'
Split rows into subsets:
componentRef:
digest: a609c3c9196484290f24a1174955f95b27f07a7b458aa5cb8cde28866cb2cb46
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Binarize column using Pandas on CSV data
type: CSV
fraction_1: '0.8'
annotations:
editor.position: '{"x":550,"y":490,"width":180,"height":40}'
Create fully connected pytorch network:
componentRef:
digest: d03d8248fd358a0275ec33568ee7dd7dce576cc112b09dfafe2651e4d97e04a9
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
output_activation_name: sigmoid
annotations:
editor.position: '{"x":380,"y":620,"width":180,"height":54}'
Create fully connected tensorflow network:
componentRef:
digest: bfcafbc5ce711b1f69cabf1338212d10d50136a73db9f9f7c984de7b80b4bfb0
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
output_activation_name: sigmoid
annotations:
editor.position: '{"x":40,"y":630,"width":180,"height":54}'
Train model using Keras on CSV:
componentRef:
digest: 42ae60c889034dbad74815653e95b4f7d576b5f47f803173e8679c7b54984609
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Create fully connected tensorflow network
type: TensorflowSavedModel
label_column_name: class
loss_function_name: binary_crossentropy
number_of_epochs: '10'
annotations:
editor.position: '{"x":40,"y":750,"width":180,"height":54}'
Train pytorch model from csv:
componentRef:
digest: 40f3185eb61e9727f41a4e0c05dd3d3b44bd802aa0f378cfc31756560033949a
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Create fully connected pytorch network
type: PyTorchScriptModule
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: class
loss_function_name: binary_cross_entropy
annotations:
editor.position: '{"x":380,"y":750,"width":180,"height":40}'
Train XGBoost model on CSV:
componentRef:
digest: 538c5a01eb38deaf532d619f0bbeaff4efc550fe1f0f776fc06791097b68ceac
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: class
objective: binary:logistic
annotations:
editor.position: '{"x":720,"y":750,"width":180,"height":40}'
Train logistic regression model using scikit learn from CSV:
componentRef:
digest: a864625a822e4b1c8ef6fe4ae1454fd90f15438f70a6712bb4c30e0dda4d35b7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/cb44b75c9c062fcc40c2b905b2024b4493dbc62b/components/ML_frameworks/Scikit_learn/Train_logistic_regression_model/from_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: class
annotations:
editor.position: '{"x":1030,"y":750,"width":180,"height":70}'
Predict with TensorFlow model on CSV data:
componentRef:
digest: 921bb1563e93a78233b8acceab87055b9154ccf5595d056028cf0396ca224cd4
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
label_column_name: class
annotations:
editor.position: '{"x":160,"y":880,"width":180,"height":54}'
Create PyTorch Model Archive with base handler:
componentRef:
digest: 8298b5ee1b0f0879f893add4cf352c8dec7cf9e21bb9db134c91a2d046cdb0ec
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml
arguments:
Model:
taskOutput:
outputName: trained_model
taskId: Train pytorch model from csv
type: PyTorchScriptModule
Model name: model
Model version: '1.0'
annotations:
editor.position: '{"x":380,"y":880,"width":180,"height":54}'
Xgboost predict on CSV:
componentRef:
digest: 0876233a0c7306fefec188bd70f059b46d1fb5aa57be231799570e3bbbdd0d95
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml
arguments:
data:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
label_column_name: class
annotations:
editor.position: '{"x":810,"y":880,"width":180,"height":40}'
Upload Scikit learn pickle model to Google Cloud Vertex AI:
componentRef:
digest: 81c91c8d7d21ec97e0872f669d68bd89edea87279d703685db54aa94743bebcd
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train logistic regression model using scikit learn from CSV
type: ScikitLearnPickleModel
annotations:
editor.position: '{"x":1030,"y":880,"width":180,"height":70}'
Upload Tensorflow model to Google Cloud Vertex AI:
componentRef:
digest: 2e45263ff640b1a688e359b6936e27a81b2407749a84f340af2aa5547e0cb92c
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
annotations:
editor.position: '{"x":40,"y":1010,"width":180,"height":54}'
Upload PyTorch model archive to Google Cloud Vertex AI:
componentRef:
digest: 4450212fae7b9001482aca7eb78b28413c205506eccf08a04e7754a8dfa99004
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model_archive:
taskOutput:
outputName: Model archive
taskId: Create PyTorch Model Archive with base handler
type: PyTorchModelArchive
annotations:
editor.position: '{"x":380,"y":1010,"width":180,"height":70}'
Upload XGBoost model to Google Cloud Vertex AI:
componentRef:
digest: 5a5a273c403670743820986c03a4175b7cb4595a556524fefcce403656286977
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
annotations:
editor.position: '{"x":720,"y":1010,"width":180,"height":54}'
outputValues: {}
@@ -1,224 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
binarize_column_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1e2558325f4c708aca75827c8acc13d230ee7e9f/components/pandas/Binarize_column/in_CSV_format/component.yaml")
split_rows_into_subsets_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml")
# TensorFlow
create_fully_connected_tensorflow_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml")
train_model_using_Keras_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml")
predict_with_TensorFlow_model_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml")
upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
# PyTorch
create_fully_connected_pytorch_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml")
train_pytorch_model_from_csv_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml")
create_pytorch_model_archive_with_base_handler_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml")
upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml")
# XGBoost
train_XGBoost_model_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml")
xgboost_predict_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml")
upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
# Scikit-learn
#train_linear_regression_model_using_scikit_learn_from_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/f807e02b54d4886c65a05f40848fd51c72407f40/components/ML_frameworks/Scikit_learn/Train_linear_regression_model/from_CSV/component.yaml")
train_logistic_regression_model_using_scikit_learn_from_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/cb44b75c9c062fcc40c2b905b2024b4493dbc62b/components/ML_frameworks/Scikit_learn/Train_logistic_regression_model/from_CSV/component.yaml")
upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
# Vertex AI
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_classification_model_using_all_frameworks_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
training_set_fraction = 0.8
# Deploying the model might incur additional costs over time
deploy_model = False
classification_label_column = "class"
all_columns = [label_column] + feature_columns
dataset = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
dataset = select_columns_using_Pandas_on_CSV_data_op(
table=dataset,
column_names=all_columns,
).outputs["transformed_table"]
dataset = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=dataset,
replacement_value="0",
# # Optional:
# column_names=None, # =[...]
).outputs["transformed_table"]
classification_dataset = binarize_column_using_Pandas_on_CSV_data_op(
table=dataset,
column_name=label_column,
predicate=" > 0",
new_column_name=classification_label_column,
).outputs["transformed_table"]
split_task = split_rows_into_subsets_op(
table=classification_dataset,
fraction_1=training_set_fraction,
)
classification_training_data = split_task.outputs["split_1"]
classification_testing_data = split_task.outputs["split_2"]
# TensorFlow
tensorflow_network = create_fully_connected_tensorflow_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
output_activation_name="sigmoid",
# output_size=1,
).outputs["model"]
tensorflow_model = train_model_using_Keras_on_CSV_op(
training_data=classification_training_data,
model=tensorflow_network,
label_column_name=classification_label_column,
# Optional:
loss_function_name="binary_crossentropy",
number_of_epochs=10,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
#metric_names=["mean_absolute_error"],
#random_seed=0,
).outputs["trained_model"]
tensorflow_predictions = predict_with_TensorFlow_model_on_CSV_data_op(
dataset=classification_testing_data,
model=tensorflow_model,
# label_column_name needs to be set when doing prediction on a dataset that has labels
label_column_name=classification_label_column,
# Optional:
# batch_size=1000,
).outputs["predictions"]
tensorflow_vertex_model_name = upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op(
model=tensorflow_model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
tensorflow_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=tensorflow_vertex_model_name,
).outputs["endpoint_name"]
# PyTorch
pytorch_network = create_fully_connected_pytorch_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
output_activation_name="sigmoid",
# output_size=1,
).outputs["model"]
pytorch_model = train_pytorch_model_from_csv_op(
model=pytorch_network,
training_data=classification_training_data,
label_column_name=classification_label_column,
loss_function_name="binary_cross_entropy",
# Optional:
#number_of_epochs=1,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
#batch_log_interval=100,
#random_seed=0,
).outputs["trained_model"]
pytorch_model_archive = create_pytorch_model_archive_with_base_handler_op(
model=pytorch_model,
# Optional:
# model_name="model",
# model_version="1.0",
).outputs["Model archive"]
pytorch_vertex_model_name = upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op(
model_archive=pytorch_model_archive,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
pytorch_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=pytorch_vertex_model_name,
).outputs["endpoint_name"]
# XGBoost
xgboost_model = train_XGBoost_model_on_CSV_op(
training_data=classification_training_data,
label_column_name=classification_label_column,
objective="binary:logistic",
# Optional:
#starting_model=None,
#num_iterations=10,
#booster_params={},
#booster="gbtree",
#learning_rate=0.3,
#min_split_loss=0,
#max_depth=6,
).outputs["model"]
# Predicting on the testing data
xgboost_predictions = xgboost_predict_on_CSV_op(
data=classification_testing_data,
model=xgboost_model,
# label_column needs to be set when doing prediction on a dataset that has labels
label_column_name=classification_label_column,
).outputs["predictions"]
xgboost_vertex_model_name = upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op(
model=xgboost_model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
xgboost_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=xgboost_vertex_model_name,
).outputs["endpoint_name"]
# Scikit-learn
sklearn_model = train_logistic_regression_model_using_scikit_learn_from_CSV_op(
dataset=classification_training_data,
label_column_name=classification_label_column,
# Optional:
#penalty="l2",
#solver="lbfgs",
#max_iterations=100,
#multi_class_mode="auto",
#random_seed=0,
).outputs["model"]
sklearn_vertex_model_name = upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op(
model=sklearn_model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
sklearn_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=sklearn_vertex_model_name,
).outputs["endpoint_name"]
pipeline_func=train_tabular_classification_model_using_all_frameworks_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,68 +0,0 @@
name: Train tabular regression linear model using Scikit learn pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_regression_linear_model_using_Scikit_learn_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":40,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":40,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":40,"y":250,"width":180,"height":54}'
Train linear regression model using scikit learn from CSV:
componentRef:
digest: c7fe7912ab0d1fb45d201d452e9ce6be5544e7d8c6d229db7a4b931ff58560f3
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/f807e02b54d4886c65a05f40848fd51c72407f40/components/ML_frameworks/Scikit_learn/Train_linear_regression_model/from_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
label_column_name: tips
annotations:
editor.position: '{"x":40,"y":360,"width":180,"height":54}'
Upload Scikit learn pickle model to Google Cloud Vertex AI:
componentRef:
digest: 81c91c8d7d21ec97e0872f669d68bd89edea87279d703685db54aa94743bebcd
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train linear regression model using scikit learn from CSV
type: ScikitLearnPickleModel
annotations:
editor.position: '{"x":40,"y":490,"width":180,"height":70}'
outputValues: {}
@@ -1,57 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
train_linear_regression_model_using_scikit_learn_from_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/f807e02b54d4886c65a05f40848fd51c72407f40/components/ML_frameworks/Scikit_learn/Train_linear_regression_model/from_CSV/component.yaml")
upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_regression_linear_model_using_Scikit_learn_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
all_columns = [label_column] + feature_columns
# Deploying the model might incur additional costs over time
deploy_model = False
training_data = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
training_data = select_columns_using_Pandas_on_CSV_data_op(
table=training_data,
column_names=all_columns,
).outputs["transformed_table"]
# Cleaning the NaN values.
training_data = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=training_data,
replacement_value="0",
#replacement_type_name="float",
).outputs["transformed_table"]
model = train_linear_regression_model_using_scikit_learn_from_CSV_op(
dataset=training_data,
label_column_name=label_column,
).outputs["model"]
vertex_model_name = upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op(
model=model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
sklearn_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func = train_tabular_regression_linear_model_using_Scikit_learn_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,97 +0,0 @@
name: Train tabular regression model using PyTorch pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_regression_model_using_PyTorch_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":240,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":240,"y":130,"width":180,"height":54}'
Create fully connected pytorch network:
componentRef:
digest: d03d8248fd358a0275ec33568ee7dd7dce576cc112b09dfafe2651e4d97e04a9
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
annotations:
editor.position: '{"x":40,"y":240,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":240,"y":240,"width":180,"height":54}'
Train pytorch model from csv:
componentRef:
digest: 40f3185eb61e9727f41a4e0c05dd3d3b44bd802aa0f378cfc31756560033949a
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Create fully connected pytorch network
type: PyTorchScriptModule
training_data:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
label_column_name: tips
annotations:
editor.position: '{"x":240,"y":380,"width":180,"height":40}'
Create PyTorch Model Archive with base handler:
componentRef:
digest: 8298b5ee1b0f0879f893add4cf352c8dec7cf9e21bb9db134c91a2d046cdb0ec
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml
arguments:
Model:
taskOutput:
outputName: trained_model
taskId: Train pytorch model from csv
type: PyTorchScriptModule
Model name: model
Model version: '1.0'
annotations:
editor.position: '{"x":240,"y":500,"width":180,"height":54}'
Upload PyTorch model archive to Google Cloud Vertex AI:
componentRef:
digest: 4450212fae7b9001482aca7eb78b28413c205506eccf08a04e7754a8dfa99004
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model_archive:
taskOutput:
outputName: Model archive
taskId: Create PyTorch Model Archive with base handler
type: PyTorchModelArchive
annotations:
editor.position: '{"x":240,"y":630,"width":180,"height":70}'
outputValues: {}
@@ -1,85 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
create_fully_connected_pytorch_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml")
train_pytorch_model_from_csv_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml")
create_pytorch_model_archive_with_base_handler_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml")
upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_regression_model_using_PyTorch_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
all_columns = [label_column] + feature_columns
# Deploying the model might incur additional costs over time
deploy_model = False
training_data = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
training_data = select_columns_using_Pandas_on_CSV_data_op(
table=training_data,
column_names=all_columns,
).outputs["transformed_table"]
# Cleaning the NaN values.
training_data = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=training_data,
replacement_value="0",
#replacement_type_name="float",
).outputs["transformed_table"]
network = create_fully_connected_pytorch_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
# output_activation_name=None,
# output_size=1,
).outputs["model"]
model = train_pytorch_model_from_csv_op(
model=network,
training_data=training_data,
label_column_name=label_column,
# Optional:
#loss_function_name="mse_loss",
#number_of_epochs=1,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
#batch_log_interval=100,
#random_seed=0,
).outputs["trained_model"]
model_archive = create_pytorch_model_archive_with_base_handler_op(
model=model,
# Optional:
# model_name="model",
# model_version="1.0",
).outputs["Model archive"]
vertex_model_name = upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op(
model_archive=model_archive,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func=train_tabular_regression_model_using_PyTorch_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,116 +0,0 @@
name: Train tabular regression model using Tensorflow pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_regression_model_using_TensorFlow_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":40,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":40,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":40,"y":250,"width":180,"height":54}'
Split rows into subsets:
componentRef:
digest: a609c3c9196484290f24a1174955f95b27f07a7b458aa5cb8cde28866cb2cb46
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
fraction_1: '0.8'
annotations:
editor.position: '{"x":170,"y":380,"width":180,"height":40}'
Create fully connected tensorflow network:
componentRef:
digest: bfcafbc5ce711b1f69cabf1338212d10d50136a73db9f9f7c984de7b80b4bfb0
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
annotations:
editor.position: '{"x":370,"y":380,"width":180,"height":54}'
Train model using Keras on CSV:
componentRef:
digest: 42ae60c889034dbad74815653e95b4f7d576b5f47f803173e8679c7b54984609
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Create fully connected tensorflow network
type: TensorflowSavedModel
label_column_name: tips
number_of_epochs: '10'
metric_names: '["mean_absolute_error"]'
annotations:
editor.position: '{"x":40,"y":500,"width":180,"height":54}'
Upload Tensorflow model to Google Cloud Vertex AI:
componentRef:
digest: 2e45263ff640b1a688e359b6936e27a81b2407749a84f340af2aa5547e0cb92c
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
annotations:
editor.position: '{"x":40,"y":630,"width":180,"height":54}'
Predict with TensorFlow model on CSV data:
componentRef:
digest: 921bb1563e93a78233b8acceab87055b9154ccf5595d056028cf0396ca224cd4
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
label_column_name: tips
annotations:
editor.position: '{"x":240,"y":630,"width":180,"height":54}'
outputValues: {}
@@ -1,97 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
split_rows_into_subsets_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml")
create_fully_connected_tensorflow_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml")
train_model_using_Keras_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml")
predict_with_TensorFlow_model_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml")
upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_regression_model_using_Tensorflow_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
training_set_fraction = 0.8
# Deploying the model might incur additional costs over time
deploy_model = False
all_columns = [label_column] + feature_columns
dataset = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
dataset = select_columns_using_Pandas_on_CSV_data_op(
table=dataset,
column_names=all_columns,
).outputs["transformed_table"]
dataset = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=dataset,
replacement_value="0",
# # Optional:
# column_names=None, # =[...]
).outputs["transformed_table"]
split_task = split_rows_into_subsets_op(
table=dataset,
fraction_1=training_set_fraction,
)
training_data = split_task.outputs["split_1"]
testing_data = split_task.outputs["split_2"]
network = create_fully_connected_tensorflow_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
# output_activation_name=None,
# output_size=1,
).outputs["model"]
model = train_model_using_Keras_on_CSV_op(
training_data=training_data,
model=network,
label_column_name=label_column,
# Optional:
#loss_function_name="mean_squared_error",
number_of_epochs=10,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
metric_names=["mean_absolute_error"],
#random_seed=0,
).outputs["trained_model"]
predictions = predict_with_TensorFlow_model_on_CSV_data_op(
dataset=testing_data,
model=model,
# label_column_name needs to be set when doing prediction on a dataset that has labels
label_column_name=label_column,
# Optional:
# batch_size=1000,
).outputs["predictions"]
vertex_model_name = upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op(
model=model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func=train_tabular_regression_model_using_Tensorflow_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,99 +0,0 @@
name: Train tabular regression model using XGBoost pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_regression_model_using_XGBoost_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":40,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":40,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":40,"y":250,"width":180,"height":54}'
Split rows into subsets:
componentRef:
digest: a609c3c9196484290f24a1174955f95b27f07a7b458aa5cb8cde28866cb2cb46
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
fraction_1: '0.8'
annotations:
editor.position: '{"x":170,"y":360,"width":180,"height":40}'
Train XGBoost model on CSV:
componentRef:
digest: 538c5a01eb38deaf532d619f0bbeaff4efc550fe1f0f776fc06791097b68ceac
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: tips
annotations:
editor.position: '{"x":40,"y":480,"width":180,"height":40}'
Upload XGBoost model to Google Cloud Vertex AI:
componentRef:
digest: 5a5a273c403670743820986c03a4175b7cb4595a556524fefcce403656286977
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
annotations:
editor.position: '{"x":40,"y":600,"width":180,"height":54}'
Xgboost predict on CSV:
componentRef:
digest: 0876233a0c7306fefec188bd70f059b46d1fb5aa57be231799570e3bbbdd0d95
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml
arguments:
data:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
label_column_name: tips
annotations:
editor.position: '{"x":240,"y":600,"width":180,"height":40}'
outputValues: {}
@@ -1,85 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
split_rows_into_subsets_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml")
train_XGBoost_model_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml")
xgboost_predict_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml")
upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_regression_model_using_XGBoost_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
training_set_fraction = 0.8
# Deploying the model might incur additional costs over time
deploy_model = False
all_columns = [label_column] + feature_columns
dataset = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
dataset = select_columns_using_Pandas_on_CSV_data_op(
table=dataset,
column_names=all_columns,
).outputs["transformed_table"]
dataset = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=dataset,
replacement_value="0",
# # Optional:
# column_names=None, # =[...]
).outputs["transformed_table"]
split_task = split_rows_into_subsets_op(
table=dataset,
fraction_1=training_set_fraction,
)
training_data = split_task.outputs["split_1"]
testing_data = split_task.outputs["split_2"]
model = train_XGBoost_model_on_CSV_op(
training_data=training_data,
label_column_name=label_column,
# Optional:
#starting_model=None,
#num_iterations=10,
#booster_params={},
#objective="reg:squarederror",
#booster="gbtree",
#learning_rate=0.3,
#min_split_loss=0,
#max_depth=6,
).outputs["model"]
# Predicting on the testing data
predictions = xgboost_predict_on_CSV_op(
data=testing_data,
model=model,
# label_column needs to be set when doing prediction on a dataset that has labels
label_column_name=label_column,
).outputs["predictions"]
vertex_model_name = upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op(
model=model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=vertex_model_name,
).outputs["endpoint_name"]
pipeline_func = train_tabular_regression_model_using_XGBoost_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
@@ -1,238 +0,0 @@
name: Train tabular regression model using all frameworks pipeline
metadata:
annotations:
author: Alexey Volkov <alexey.volkov@ark-kun.com>
canonical_location: https://raw.githubusercontent.com/Ark-kun/pipeline_components/master/samples/Google_Cloud_Vertex_AI/Train_tabular_regression_model_using_all_frameworks_and_import_to_Vertex_AI/pipeline.component.yaml
sdk: https://cloud-pipelines.net/pipeline-editor/
implementation:
graph:
tasks:
Download from GCS:
componentRef:
digest: 4175c9ff143cb8cc75d05451c0a0ebdf5a0d6d020816e29f5e9cefbb7d56f241
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
GCS path: gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv
annotations:
editor.position: '{"x":550,"y":40,"width":180,"height":40}'
Select columns using Pandas on CSV data:
componentRef:
digest: 9b9500f461c1d04f1e48992de9138db14a6800f23649d73048673d5ea6dc56ad
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: Data
taskId: Download from GCS
column_names: '["tips", "trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"]'
annotations:
editor.position: '{"x":550,"y":140,"width":180,"height":54}'
Fill all missing values using Pandas on CSV data:
componentRef:
digest: a1b0c29a4615f2e3652aa5d31b9255fa15700e146627c755f8fc172f82e71af7
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Select columns using Pandas on CSV data
type: CSV
replacement_value: '0'
annotations:
editor.position: '{"x":550,"y":250,"width":180,"height":54}'
Split rows into subsets:
componentRef:
digest: a609c3c9196484290f24a1174955f95b27f07a7b458aa5cb8cde28866cb2cb46
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml
arguments:
table:
taskOutput:
outputName: transformed_table
taskId: Fill all missing values using Pandas on CSV data
type: CSV
fraction_1: '0.8'
annotations:
editor.position: '{"x":550,"y":360,"width":180,"height":40}'
Create fully connected pytorch network:
componentRef:
digest: d03d8248fd358a0275ec33568ee7dd7dce576cc112b09dfafe2651e4d97e04a9
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
annotations:
editor.position: '{"x":380,"y":490,"width":180,"height":54}'
Create fully connected tensorflow network:
componentRef:
digest: bfcafbc5ce711b1f69cabf1338212d10d50136a73db9f9f7c984de7b80b4bfb0
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml
arguments:
input_size: '7'
hidden_layer_sizes: '[10]'
activation_name: elu
annotations:
editor.position: '{"x":40,"y":500,"width":180,"height":54}'
Train model using Keras on CSV:
componentRef:
digest: 42ae60c889034dbad74815653e95b4f7d576b5f47f803173e8679c7b54984609
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Create fully connected tensorflow network
type: TensorflowSavedModel
label_column_name: tips
number_of_epochs: '10'
metric_names: '["mean_absolute_error"]'
annotations:
editor.position: '{"x":40,"y":620,"width":180,"height":54}'
Train pytorch model from csv:
componentRef:
digest: 40f3185eb61e9727f41a4e0c05dd3d3b44bd802aa0f378cfc31756560033949a
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Create fully connected pytorch network
type: PyTorchScriptModule
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: tips
annotations:
editor.position: '{"x":380,"y":620,"width":180,"height":40}'
Train XGBoost model on CSV:
componentRef:
digest: 538c5a01eb38deaf532d619f0bbeaff4efc550fe1f0f776fc06791097b68ceac
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml
arguments:
training_data:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: tips
annotations:
editor.position: '{"x":720,"y":620,"width":180,"height":40}'
Train linear regression model using scikit learn from CSV:
componentRef:
digest: c7fe7912ab0d1fb45d201d452e9ce6be5544e7d8c6d229db7a4b931ff58560f3
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/f807e02b54d4886c65a05f40848fd51c72407f40/components/ML_frameworks/Scikit_learn/Train_linear_regression_model/from_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: split_1
taskId: Split rows into subsets
type: CSV
label_column_name: tips
annotations:
editor.position: '{"x":1030,"y":620,"width":180,"height":54}'
Predict with TensorFlow model on CSV data:
componentRef:
digest: 921bb1563e93a78233b8acceab87055b9154ccf5595d056028cf0396ca224cd4
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml
arguments:
dataset:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
label_column_name: tips
annotations:
editor.position: '{"x":160,"y":750,"width":180,"height":54}'
Create PyTorch Model Archive with base handler:
componentRef:
digest: 8298b5ee1b0f0879f893add4cf352c8dec7cf9e21bb9db134c91a2d046cdb0ec
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml
arguments:
Model:
taskOutput:
outputName: trained_model
taskId: Train pytorch model from csv
type: PyTorchScriptModule
Model name: model
Model version: '1.0'
annotations:
editor.position: '{"x":380,"y":750,"width":180,"height":54}'
Xgboost predict on CSV:
componentRef:
digest: 0876233a0c7306fefec188bd70f059b46d1fb5aa57be231799570e3bbbdd0d95
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml
arguments:
data:
taskOutput:
outputName: split_2
taskId: Split rows into subsets
type: CSV
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
label_column_name: tips
annotations:
editor.position: '{"x":810,"y":750,"width":180,"height":40}'
Upload Scikit learn pickle model to Google Cloud Vertex AI:
componentRef:
digest: 81c91c8d7d21ec97e0872f669d68bd89edea87279d703685db54aa94743bebcd
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train linear regression model using scikit learn from CSV
type: ScikitLearnPickleModel
annotations:
editor.position: '{"x":1030,"y":750,"width":180,"height":70}'
Upload Tensorflow model to Google Cloud Vertex AI:
componentRef:
digest: 2e45263ff640b1a688e359b6936e27a81b2407749a84f340af2aa5547e0cb92c
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: trained_model
taskId: Train model using Keras on CSV
type: TensorflowSavedModel
annotations:
editor.position: '{"x":40,"y":880,"width":180,"height":54}'
Upload PyTorch model archive to Google Cloud Vertex AI:
componentRef:
digest: 4450212fae7b9001482aca7eb78b28413c205506eccf08a04e7754a8dfa99004
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model_archive:
taskOutput:
outputName: Model archive
taskId: Create PyTorch Model Archive with base handler
type: PyTorchModelArchive
annotations:
editor.position: '{"x":380,"y":880,"width":180,"height":70}'
Upload XGBoost model to Google Cloud Vertex AI:
componentRef:
digest: 5a5a273c403670743820986c03a4175b7cb4595a556524fefcce403656286977
url: https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml
arguments:
model:
taskOutput:
outputName: model
taskId: Train XGBoost model on CSV
type: XGBoostModel
annotations:
editor.position: '{"x":720,"y":880,"width":180,"height":54}'
outputValues: {}
@@ -1,208 +0,0 @@
# python3 -m pip install "kfp<2.0.0" "google-cloud-aiplatform>=1.16.0" --upgrade --quiet
from kfp import components
# %% Loading components
download_from_gcs_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/storage/download/workaround_for_buggy_KFPv2_compiler/component.yaml")
select_columns_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/8c78aae096806cff3bc331a40566f42f5c3e9d4b/components/pandas/Select_columns/in_CSV_format/component.yaml")
fill_all_missing_values_using_Pandas_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/23405971f5f16a41b16c343129b893c52e4d1d48/components/pandas/Fill_all_missing_values/in_CSV_format/component.yaml")
split_rows_into_subsets_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/daae5a4abaa35e44501818b1534ed7827d7da073/components/dataset_manipulation/Split_rows_into_subsets/in_CSV/component.yaml")
# TensorFlow
create_fully_connected_tensorflow_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/9ca0f9eecf5f896f65b8538bbd809747052617d1/components/tensorflow/Create_fully_connected_network/component.yaml")
train_model_using_Keras_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c504a4010348c50eaaf6d4337586ccc008f4dcef/components/tensorflow/Train_model_using_Keras/on_CSV/component.yaml")
predict_with_TensorFlow_model_on_CSV_data_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/59c759ce6f543184e30db6817d2a703879bc0f39/components/tensorflow/Predict/on_CSV/component.yaml")
upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Tensorflow_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
# PyTorch
create_fully_connected_pytorch_network_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/1a2ef3eeb77bc278f33cad0dd29008ea2431e191/components/PyTorch/Create_fully_connected_network/component.yaml")
train_pytorch_model_from_csv_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/d8c4cf5e6403bc65bcf8d606e6baf87e2528a3dc/components/PyTorch/Train_PyTorch_model/from_CSV/component.yaml")
create_pytorch_model_archive_with_base_handler_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/46d51383e6554b7f3ab4fd8cf614d8c2b422fb22/components/PyTorch/Create_PyTorch_Model_Archive/with_base_handler/component.yaml")
upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_PyTorch_model_archive/workaround_for_buggy_KFPv2_compiler/component.yaml")
# XGBoost
train_XGBoost_model_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/58d3a47f904f32a64af8403330ba7e2134cae46d/components/XGBoost/Train/component.yaml")
xgboost_predict_on_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/4694ec97baccf59284c2a1db4aa2250c22291eab/components/XGBoost/Predict/component.yaml")
upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_XGBoost_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
# Scikit-learn
train_linear_regression_model_using_scikit_learn_from_CSV_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/f807e02b54d4886c65a05f40848fd51c72407f40/components/ML_frameworks/Scikit_learn/Train_linear_regression_model/from_CSV/component.yaml")
upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/c6a8b67d1ada2cc17665c99ff6b410df588bee28/components/google-cloud/Vertex_AI/Models/Upload_Scikit-learn_pickle_model/workaround_for_buggy_KFPv2_compiler/component.yaml")
# Vertex AI
deploy_model_to_endpoint_op = components.load_component_from_url("https://raw.githubusercontent.com/Ark-kun/pipeline_components/27a5ea25e849c9e8c0cb6ed65518bc3ece259aaf/components/google-cloud/Vertex_AI/Models/Deploy_to_endpoint/workaround_for_buggy_KFPv2_compiler/component.yaml")
# %% Pipeline definition
def train_tabular_regression_model_using_all_frameworks_pipeline():
dataset_gcs_uri = "gs://ml-pipeline-dataset/Chicago_taxi_trips/chicago_taxi_trips_2019-01-01_-_2019-02-01_limit=10000.csv"
feature_columns = ["trip_seconds", "trip_miles", "pickup_community_area", "dropoff_community_area", "fare", "tolls", "extras"] # Excluded "trip_total"
label_column = "tips"
training_set_fraction = 0.8
# Deploying the model might incur additional costs over time
deploy_model = False
all_columns = [label_column] + feature_columns
dataset = download_from_gcs_op(
gcs_path=dataset_gcs_uri
).outputs["Data"]
dataset = select_columns_using_Pandas_on_CSV_data_op(
table=dataset,
column_names=all_columns,
).outputs["transformed_table"]
dataset = fill_all_missing_values_using_Pandas_on_CSV_data_op(
table=dataset,
replacement_value="0",
# # Optional:
# column_names=None, # =[...]
).outputs["transformed_table"]
split_task = split_rows_into_subsets_op(
table=dataset,
fraction_1=training_set_fraction,
)
training_data = split_task.outputs["split_1"]
testing_data = split_task.outputs["split_2"]
# TensorFlow
tensorflow_network = create_fully_connected_tensorflow_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
# output_activation_name=None,
# output_size=1,
).outputs["model"]
tensorflow_model = train_model_using_Keras_on_CSV_op(
training_data=training_data,
model=tensorflow_network,
label_column_name=label_column,
# Optional:
#loss_function_name="mean_squared_error",
number_of_epochs=10,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
metric_names=["mean_absolute_error"],
#random_seed=0,
).outputs["trained_model"]
tensorflow_predictions = predict_with_TensorFlow_model_on_CSV_data_op(
dataset=testing_data,
model=tensorflow_model,
# label_column_name needs to be set when doing prediction on a dataset that has labels
label_column_name=label_column,
# Optional:
# batch_size=1000,
).outputs["predictions"]
tensorflow_vertex_model_name = upload_Tensorflow_model_to_Google_Cloud_Vertex_AI_op(
model=tensorflow_model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
tensorflow_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=tensorflow_vertex_model_name,
).outputs["endpoint_name"]
# PyTorch
pytorch_network = create_fully_connected_pytorch_network_op(
input_size=len(feature_columns),
# Optional:
hidden_layer_sizes=[10],
activation_name="elu",
# output_activation_name=None,
# output_size=1,
).outputs["model"]
pytorch_model = train_pytorch_model_from_csv_op(
model=pytorch_network,
training_data=training_data,
label_column_name=label_column,
# Optional:
#loss_function_name="mse_loss",
#number_of_epochs=1,
#learning_rate=0.1,
#optimizer_name="Adadelta",
#optimizer_parameters={},
#batch_size=32,
#batch_log_interval=100,
#random_seed=0,
).outputs["trained_model"]
pytorch_model_archive = create_pytorch_model_archive_with_base_handler_op(
model=pytorch_model,
# Optional:
# model_name="model",
# model_version="1.0",
).outputs["Model archive"]
pytorch_vertex_model_name = upload_PyTorch_model_archive_to_Google_Cloud_Vertex_AI_op(
model_archive=pytorch_model_archive,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
pytorch_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=pytorch_vertex_model_name,
).outputs["endpoint_name"]
# XGBoost
xgboost_model = train_XGBoost_model_on_CSV_op(
training_data=training_data,
label_column_name=label_column,
# Optional:
#starting_model=None,
#num_iterations=10,
#booster_params={},
#objective="reg:squarederror",
#booster="gbtree",
#learning_rate=0.3,
#min_split_loss=0,
#max_depth=6,
).outputs["model"]
# Predicting on the testing data
xgboost_predictions = xgboost_predict_on_CSV_op(
data=testing_data,
model=xgboost_model,
# label_column needs to be set when doing prediction on a dataset that has labels
label_column_name=label_column,
).outputs["predictions"]
xgboost_vertex_model_name = upload_XGBoost_model_to_Google_Cloud_Vertex_AI_op(
model=xgboost_model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
xgboost_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=xgboost_vertex_model_name,
).outputs["endpoint_name"]
# Scikit-learn
sklearn_model = train_linear_regression_model_using_scikit_learn_from_CSV_op(
dataset=training_data,
label_column_name=label_column,
).outputs["model"]
sklearn_vertex_model_name = upload_Scikit_learn_pickle_model_to_Google_Cloud_Vertex_AI_op(
model=sklearn_model,
).outputs["model_name"]
# Deploying the model might incur additional costs over time
if deploy_model:
sklearn_vertex_endpoint_name = deploy_model_to_endpoint_op(
model_name=sklearn_vertex_model_name,
).outputs["endpoint_name"]
pipeline_func=train_tabular_regression_model_using_all_frameworks_pipeline
# %% Pipeline submission
if __name__ == '__main__':
from google.cloud import aiplatform
aiplatform.PipelineJob.from_pipeline_func(pipeline_func=pipeline_func).submit()
+1 -2
View File
@@ -12,6 +12,7 @@
/managed_notebooks/
/bigquery_ml/ @polong
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
/pipelines/google_cloud_pipeline_components_TPU_model_train_upload_deploy.ipynb @brianchunkang
/explainable_ai/SDK_Custom_Container_XAI.ipynb @brianchunkang
/matching_engine/sdk_matching_engine_for_indexing.ipynb @ivanmkc
@@ -22,7 +23,6 @@
/feature_store @nayaknishant @morgandu
/prediction @googleapis/vertex-prediction-team
/vertex_endpoints/tf_hub_obj_detection/deploy_tfhub_object_detection_on_vertex_endpoints.ipynb @entrpn
/vertex_endpoints/find_ideal_machine_type/find_ideal_machine_type/find_ideal_machine_type.ipynb @entrpn
/vertex_endpoints/nvidia-triton/nvidia-triton-custom-container-prediction.ipynb @RajeshThallam
/vertex_endpoints/optimized_tensorflow_runtime @vlasenkoalexey
/notebooks/community/ml_ops/stage2/get_started_with_visionapi_and_automl.ipynb @mansari
@@ -34,4 +34,3 @@
/notebooks/community/vertex-ai-samples/notebooks/community/model_registry/vertex_ai_model_registry_bqml_custom_model_versioning.ipynb @inardini
/notebooks/community/vertex-ai-samples/notebooks/community/model_registry/vertex_ai_model_registry_automl_model_versioning.ipynb @inardini
/notebooks/community/vizier/conversions_vertex_vizier_and_open_source_vizier.ipynb @halio-g
/notebooks/community/experiments/vertex_ai_model_experimentation.ipynb @inardini @asobran
File diff suppressed because it is too large Load Diff
@@ -1,929 +0,0 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "copyright"
},
"outputs": [],
"source": [
"# Copyright 2022 Google LLC\n",
"#\n",
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
"# you may not use this file except in compliance with the License.\n",
"# You may obtain a copy of the License at\n",
"#\n",
"# https://www.apache.org/licenses/LICENSE-2.0\n",
"#\n",
"# Unless required by applicable law or agreed to in writing, software\n",
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
"# See the License for the specific language governing permissions and\n",
"# limitations under the License."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "title"
},
"source": [
"# E2E ML on GCP: MLOps stage 2 : Get started with autologging using Vertex AI Experiments for XGBoost models\n",
"\n",
"<table align=\"left\">\n",
" <td>\n",
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_with_vertex_experiments_autologging_xgboost.ipynb\">\n",
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/ml_ops/stage2/get_started_with_vertex_experiments_autologging_xgboost.ipynb\">\n",
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
" View on GitHub\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/ml_ops/stage2/get_started_with_vertex_experiments_autologging_xgboost.ipynb\">\n",
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
" Open in Vertex AI Workbench\n",
" </a>\n",
" </td>\n",
"</table>\n",
"<br/><br/><br/>"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "overview:automl"
},
"source": [
"## Overview\n",
"\n",
"\n",
"This tutorial demonstrates how to use the `Vertex AI Experiments` with DIY code to implement automatic logging of parameters and metrics for experiments."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "objective:automl,training,batch_prediction"
},
"source": [
"### Objective\n",
"\n",
"In this tutorial, you learn how to create an experiment for training an XGBoost model, and automatically log parameters and metrics using the enclosed do-it-yourself (DIY) code.\n",
"\n",
"This tutorial uses the following Google Cloud ML services and resources:\n",
"\n",
"- `Vertex AI Experiments`\n",
"\n",
"The steps performed include:\n",
"\n",
"- Construct the DIY autologging code.\n",
"- Construct training package with call to autologging.\n",
"- Train a model.\n",
"- View the experiment\n",
"- Delete the experiment."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "dataset:custom,boston,lrg"
},
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "costs"
},
"source": [
"### Costs\n",
"\n",
"This tutorial uses billable components of Google Cloud:\n",
"\n",
"* Vertex AI\n",
"* Cloud Storage\n",
"\n",
"Learn about [Vertex AI\n",
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
"Calculator](https://cloud.google.com/products/calculator/)\n",
"to generate a cost estimate based on your projected usage."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "setup_local"
},
"source": [
"### Set up your local development environment\n",
"\n",
"If you are using Colab or Vertex Workbench AI Notebooks, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"\n",
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
"\n",
"- The Cloud Storage SDK\n",
"- Git\n",
"- Python 3\n",
"- virtualenv\n",
"- Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"The Cloud Storage guide to [Setting up a Python development environment](https://cloud.google.com/python/setup) and the [Jupyter installation guide](https://jupyter.org/install) provide detailed instructions for meeting these requirements. The following steps provide a condensed set of instructions:\n",
"\n",
"1. [Install and initialize the SDK](https://cloud.google.com/sdk/docs/).\n",
"\n",
"2. [Install Python 3](https://cloud.google.com/python/setup#installing_python).\n",
"\n",
"3. [Install virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv) and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"4. To install Jupyter, run `pip3 install jupyter` on the command-line in a terminal shell.\n",
"\n",
"5. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"6. Open this notebook in the Jupyter Notebook Dashboard.\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_aip:mbsdk"
},
"source": [
"## Installation\n",
"\n",
"Install the following packages to execute this notebook."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_aip:mbsdk"
},
"outputs": [],
"source": [
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install --upgrade --quiet {USER_FLAG} google-cloud-aiplatform \\\n",
" xgboost \\\n",
" scikit-learn \\\n",
" numpy"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "restart"
},
"source": [
"### Restart the kernel\n",
"\n",
"After you install the additional packages, you need to restart the notebook kernel so it can find the packages."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "restart"
},
"outputs": [],
"source": [
"import os\n",
"\n",
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "before_you_begin:nogpu"
},
"source": [
"## Before you begin\n",
"\n",
"### GPU runtime\n",
"\n",
"This tutorial does not require a GPU runtime.\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
"\n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"4. If you are running this notebook locally, you need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "project_id"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_project_id"
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_project_id"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_gcloud_project_id"
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "region"
},
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "region"
},
"outputs": [],
"source": [
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "timestamp"
},
"source": [
"#### UUID\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "timestamp"
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "gcp_authenticate"
},
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated. \n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"**Click Create service account**.\n",
"\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "gcp_authenticate"
},
"outputs": [],
"source": [
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "setup_vars"
},
"source": [
"### Import libraries and define constants"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "59963fb7178f"
},
"outputs": [],
"source": [
"import google.cloud.aiplatform as aiplatform\n",
"import numpy as np\n",
"import xgboost as xgb\n",
"from sklearn.metrics import accuracy_score, precision_score, recall_score\n",
"\n",
"# to suppress lint message (unused)\n",
"precision_score, recall_score"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "init_aip:mbsdk"
},
"source": [
"## Initialize Vertex AI SDK for Python\n",
"\n",
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "init_aip:mbsdk"
},
"outputs": [],
"source": [
"aiplatform.init(project=PROJECT_ID, location=REGION)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ae8f31c8c617"
},
"source": [
"## DIY code for autologging XGBoost models\n",
"\n",
"The code below implements autologging for XGBoost models.\n",
"\n",
"- `autologging()`: Initializes the experiment and uses heap injection to replace `xgboost.train()` symbols on the heap with the redirect wrapper function `VertexXGBtrain`.\n",
"\n",
"- `VertexXGBtrain`: A wrapper function for XGBoost train() function. Automatically logs hyperparameters and calls the underlyig function.\n",
"\n",
"- `VertexSKLaccuracy_score`: A wrapper function for scikit-learn accuracy_score() function. Automatically calls underlying function and logs the metrics results."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "8eb012e5d7ef"
},
"outputs": [],
"source": [
"def autolog(\n",
" project: str = None,\n",
" location: str = None,\n",
" staging_bucket: str = None,\n",
" experiment: str = None,\n",
" run: str = None,\n",
" framework: str = \"tf\",\n",
"):\n",
" \"\"\"\n",
" Enable automatic logging of parameters and metrics in Vertex AI Experiments,\n",
" for corresponding framework.\n",
"\n",
" project: The project ID\n",
" location : The region\n",
" staging_bucket: temporary bucket\n",
" experiment: The name of the experiment\n",
" run: The name of the run within the experiment\n",
" framework: The ML framework for which a model is being trained.\n",
" \"\"\"\n",
" # autologging\n",
" if framework == \"tf\":\n",
" try:\n",
" globals()[\"Sequential\"] = VertexTFSequential\n",
" if \"tf\" in globals():\n",
" tf.keras.Sequential = VertexTFSequential\n",
" if \"tensorflow\" in globals():\n",
" tensorflow.keras.Sequential = VertexTFSequential\n",
" except:\n",
" pass\n",
"\n",
" try:\n",
" globals()[\"Model\"] = VertexTFModel\n",
" if \"tf\" in globals():\n",
" tf.keras.Model = VertexTFModel\n",
" if \"tensorflow\" in globals():\n",
" tensorflow.keras.Model = VertexTFModel\n",
" except:\n",
" pass\n",
" elif framework == \"xgb\":\n",
" global real_xgb_train\n",
" global real_accuracy_score, real_precision_score, real_recall_score\n",
" import sklearn\n",
"\n",
" try:\n",
" if \"xgboost\" in globals():\n",
" real_xgb_train = xgboost.train\n",
" xgboost.train = VertexXGBtrain\n",
" except:\n",
" pass\n",
"\n",
" try:\n",
" if \"xgb\" in globals():\n",
" real_xgb_train = xgb.train\n",
" xgb.train = VertexXGBtrain\n",
" except:\n",
" pass\n",
"\n",
" try:\n",
" global accuracy_score, precision_score, recall_score\n",
" if \"accuracy_score\" in globals():\n",
" real_accuracy_score = sklearn.metrics.accuracy_score\n",
" sklearn.metrics.accuracy_score = VertexSKLaccuracy_score\n",
" accuracy_score = VertexSKLaccuracy_score\n",
" if \"precision_score\" in globals():\n",
" real_precision_score = sklearn.metrics.precision_score\n",
" sklearn.metrics.precision_score = VertexSKLprecision_score\n",
" precision_score = VertexSKLprecision_score\n",
" if \"recall_score\" in globals():\n",
" real_recall_score = sklearn.metrics.recall_score\n",
" sklearn.metrics.recall_score = VertexSKLrecall_score\n",
" recall_score = VertexSKLrecall_score\n",
" except:\n",
" pass\n",
"\n",
" if project:\n",
" aiplatform.init(\n",
" project=project, location=location, staging_bucket=staging_bucket\n",
" )\n",
"\n",
" if experiment:\n",
" aiplatform.init(experiment=experiment)\n",
" if run:\n",
" aiplatform.start_run(run)\n",
"\n",
"\n",
"def VertexXGBtrain(\n",
" params,\n",
" dtrain,\n",
" num_boost_round=10,\n",
" evals=None,\n",
" obj=None,\n",
" maximize=None,\n",
" early_stopping_rounds=None,\n",
" evals_result=None,\n",
" verbose_eval=True,\n",
" callbacks=None,\n",
" custom_metric=None,\n",
"):\n",
" \"\"\"\n",
" Wrapper function for autologging training parameters with Vertex AI Experiments\n",
" Args:\n",
" same as underlying xgb.train() method\n",
" \"\"\"\n",
" global real_xgb_train\n",
"\n",
" aiplatform.log_params({\"train.num_boost_round\": int(num_boost_round)})\n",
"\n",
" if params:\n",
" if \"booster\" in params:\n",
" aiplatform.log_params({\"train.booster\": int(params[\"booster\"])})\n",
"\n",
" # booster parameters\n",
" if \"eta\" in params:\n",
" aiplatform.log_params({\"train.eta\": int(params[\"eta\"])})\n",
" if \"max_depth\" in params:\n",
" aiplatform.log_params({\"train.max_depth\": int(params[\"max_depth\"])})\n",
" if \"max_leaf_nodes\" in params:\n",
" aiplatform.log_params(\n",
" {\"train.max_leaf_nodes\": int(params[\"max_leaf_nodes\"])}\n",
" )\n",
" if \"gamma\" in params:\n",
" aiplatform.log_params({\"train.gamma\": int(params[\"gamma\"])})\n",
" if \"alpha\" in params:\n",
" aiplatform.log_params({\"train.alpha\": int(params[\"alpha\"])})\n",
"\n",
" return real_xgb_train(\n",
" params=params,\n",
" dtrain=dtrain,\n",
" num_boost_round=num_boost_round,\n",
" evals=evals,\n",
" obj=obj,\n",
" maximize=maximize,\n",
" early_stopping_rounds=early_stopping_rounds,\n",
" evals_result=evals_result,\n",
" verbose_eval=verbose_eval,\n",
" callbacks=callbacks,\n",
" custom_metric=custom_metric,\n",
" )\n",
"\n",
"\n",
"def VertexSKLaccuracy_score(labels, predictions):\n",
" \"\"\"\n",
" Wrapper function for autologging training metrics with Vertex AI Experiments\n",
" Args:\n",
" same as underlying accuracy_score function\n",
" \"\"\"\n",
" global real_accuracy_score\n",
" accuracy = real_accuracy_score(labels, predictions)\n",
" aiplatform.log_metrics({\"accuracy\": accuracy})\n",
" return accuracy\n",
"\n",
"\n",
"def VertexSKLprecision_score(\n",
" y_true,\n",
" y_pred,\n",
" *,\n",
" labels=None,\n",
" pos_label=1,\n",
" average=\"binary\",\n",
" sample_weight=None,\n",
" zero_division=\"warn\",\n",
"):\n",
" \"\"\"\n",
" Wrapper function for autologging training metrics with Vertex AI Experiments\n",
" Args:\n",
" same as underlying precision_score function\n",
" \"\"\"\n",
" global real_precision_score\n",
" precision = real_precision_score(\n",
" y_true,\n",
" y_pred,\n",
" labels=labels,\n",
" pos_label=pos_label,\n",
" average=average,\n",
" sample_weight=sample_weight,\n",
" zero_division=zero_division,\n",
" )\n",
" aiplatform.log_metrics({\"precision\": precision})\n",
" return precision\n",
"\n",
"\n",
"def VertexSKLrecall_score(\n",
" y_true,\n",
" y_pred,\n",
" *,\n",
" labels=None,\n",
" pos_label=1,\n",
" average=\"binary\",\n",
" sample_weight=None,\n",
" zero_division=\"warn\",\n",
"):\n",
" \"\"\"\n",
" Wrapper function for autologging training metrics with Vertex AI Experiments\n",
" Args:\n",
" same as underlying recall_score function\n",
" \"\"\"\n",
" global real_recall_score\n",
" recall = real_recall_score(\n",
" y_true,\n",
" y_pred,\n",
" labels=labels,\n",
" pos_label=pos_label,\n",
" average=average,\n",
" sample_weight=sample_weight,\n",
" zero_division=zero_division,\n",
" )\n",
" aiplatform.log_metrics({\"recall\": recall})\n",
" return recall\n",
"\n",
"\n",
"class VertexXGBBooster(xgb.Booster):\n",
" \"\"\"\n",
" WIP\n",
" \"\"\"\n",
"\n",
" def __init__(self, params=None, cache=None, model_file=None):\n",
" super().__init__(params, cache, model_file)\n",
"\n",
" def boost(\n",
" self, dtrain: xgb.core.DMatrix, grad: np.ndarray, hess: np.ndarray\n",
" ) -> None:\n",
" return super().boost(dtrain, grad, hess)\n",
"\n",
" def eval(\n",
" self, data: xgb.core.DMatrix, name: str = \"eval\", iteration: int = 0\n",
" ) -> str:\n",
" return super().eval(data, name, iteration)\n",
"\n",
" def update(self, dtrain: xgb.core.DMatrix, iteration: int, fobj=None) -> None:\n",
" return super().update(dtrain, iteration, fobj)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ce76826902c0"
},
"source": [
"### Train the XGBoost model with Vertex AI Experiments\n",
"\n",
"In the following code, you build, train and evaluate an XGBoost tabular model. The Python script includes the following calls to integrate `Vertex AI Experiments`:\n",
"\n",
"- command-line arguments: The arguments `experiment` and `run` are used to pass in the experiment and run names for the experiment.\n",
"- `autologging()`: Initializes the experiment and does the heap injection.\n",
"- `aiplatform.start_execution()`: Initializes a context for linking artifacts.\n",
"- `aiplatform.end_run()`: Ends the experiment.\n",
"\n",
"*Note:* The functions `xgb.train` and `accuracy_score` will be redirected to `VertexXGBtrain` and VertexSKLaccuracy_score, respectively, by heap injection. When subsequent calls are made to the `train()` and `accuracy()` function,s they will be executed as the corresponding `VertexXGBtrain` and `VertexSKLaccuracy_score` functions."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "WiSnFuDoox9W"
},
"outputs": [],
"source": [
"EXPERIMENT_NAME = f\"myexperiment{UUID}\"\n",
"RUN_NAME = \"run-1\"\n",
"\n",
"DATASET_DIR = \"gs://cloud-samples-data/ai-platform/iris\"\n",
"DATASET_DATA_URL = DATASET_DIR + \"/iris_data.csv\"\n",
"DATASET_LABELS_URL = DATASET_DIR + \"/iris_target.csv\"\n",
"\n",
"BOOSTED_ROUNDS = 20\n",
"\n",
"import logging\n",
"import os\n",
"import subprocess\n",
"import sys\n",
"\n",
"import hypertune\n",
"import numpy as np\n",
"import pandas as pd\n",
"import xgboost as xgb\n",
"from sklearn.model_selection import train_test_split\n",
"\n",
"\n",
"def get_data():\n",
" # gsutil outputs everything to stderr so we need to divert it to stdout.\n",
" subprocess.check_call(\n",
" [\"gsutil\", \"cp\", DATASET_DATA_URL, \"data.csv\"], stderr=sys.stdout\n",
" )\n",
" # gsutil outputs everything to stderr so we need to divert it to stdout.\n",
" subprocess.check_call(\n",
" [\"gsutil\", \"cp\", DATASET_LABELS_URL, \"labels.csv\"], stderr=sys.stdout\n",
" )\n",
"\n",
" # Load data into pandas, then use `.values` to get NumPy arrays\n",
" data = pd.read_csv(\"data.csv\").values\n",
" labels = pd.read_csv(\"labels.csv\").values\n",
"\n",
" # Convert one-column 2D array into 1D array for use with XGBoost\n",
" labels = labels.reshape((labels.size,))\n",
"\n",
" train_data, test_data, train_labels, test_labels = train_test_split(\n",
" data, labels, test_size=0.2, random_state=7\n",
" )\n",
"\n",
" # Load data into DMatrix object\n",
" dtrain = xgb.DMatrix(train_data, label=train_labels)\n",
" return dtrain, test_data, test_labels\n",
"\n",
"\n",
"def train_model(dtrain):\n",
" logging.info(\"Start training ...\")\n",
" # Train XGBoost model\n",
" params = {\"max_depth\": 3, \"objective\": \"multi:softmax\", \"num_class\": 3}\n",
" model = xgb.train(params=params, dtrain=dtrain, num_boost_round=BOOSTED_ROUNDS)\n",
" logging.info(\"Training completed\")\n",
" return model\n",
"\n",
"\n",
"def evaluate_model(model, test_data, test_labels):\n",
" dtest = xgb.DMatrix(test_data)\n",
" pred = model.predict(dtest)\n",
" predictions = [round(value) for value in pred]\n",
" # evaluate predictions\n",
" accuracy = accuracy_score(test_labels, predictions)\n",
"\n",
" logging.info(f\"Evaluation completed with model accuracy: {accuracy}\")\n",
"\n",
" # report metric for hyperparameter tuning\n",
" hpt = hypertune.HyperTune()\n",
" hpt.report_hyperparameter_tuning_metric(\n",
" hyperparameter_metric_tag=\"accuracy\", metric_value=accuracy\n",
" )\n",
" return accuracy\n",
"\n",
"\n",
"# autologging\n",
"autolog(experiment=EXPERIMENT_NAME, run=RUN_NAME, framework=\"xgb\")\n",
"\n",
"with aiplatform.start_execution(\n",
" schema_title=\"system.ContainerExecution\", display_name=\"example_training\"\n",
") as execution:\n",
" dtrain, test_data, test_labels = get_data()\n",
" model = train_model(dtrain)\n",
" accuracy = evaluate_model(model, test_data, test_labels)\n",
"\n",
"aiplatform.end_run()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "5f40912e6500"
},
"source": [
"#### Get the experiment results\n",
"\n",
"Next, you use the experiment name as a parameter to the method `get_experiment_df()` to get the results of the experiment as a pandas dataframe."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "7e9671712230"
},
"outputs": [],
"source": [
"experiment_df = aiplatform.get_experiment_df()\n",
"experiment_df = experiment_df[experiment_df.experiment_name == EXPERIMENT_NAME]\n",
"experiment_df.T"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "e508c159d712"
},
"source": [
"#### Delete the experiment\n",
"\n",
"Since the experiment was created within a training script, to delete the experiment you use the `list()` method to obtain all the experiments for the project, and then filter on the experiment name."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "1a1b5fcbfde0"
},
"outputs": [],
"source": [
"experiments = aiplatform.Experiment.list()\n",
"for experiment in experiments:\n",
" if experiment.name == EXPERIMENT_NAME:\n",
" experiment.delete()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "cleanup:mbsdk"
},
"source": [
"# Cleaning up\n",
"\n",
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
"\n",
"Otherwise, you can delete the individual resources you created in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "9eb897e0e728"
},
"outputs": [],
"source": [
"# There are no resources to cleanup"
]
}
],
"metadata": {
"colab": {
"name": "get_started_with_vertex_experiments_autologging_xgboost.ipynb",
"toc_visible": true
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
File diff suppressed because it is too large Load Diff
+372 -101
View File
@@ -31,8 +31,6 @@
"source": [
"# [TODO] Add your H1 title heading here\n",
"\n",
"{TODO: Update the links below.} \n",
"\n",
"<table align=\"left\">\n",
"\n",
" <td>\n",
@@ -55,17 +53,6 @@
"</table>"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "24743cf4a1e1"
},
"source": [
"**_NOTE_**: This notebook has been tested in the following environment:\n",
"\n",
"* Python version = 3.9"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -124,7 +111,7 @@
"This tutorial uses billable components of Google Cloud:\n",
"\n",
"* Vertex AI\n",
"* {TODO: BigQuery}\n",
"* {TODO: BigQyuery}\n",
"* Cloud Storage\n",
"\n",
"{TODO: Include links to pricing documentation for each product you listed above.\n",
@@ -138,6 +125,62 @@
"to generate a cost estimate based on your projected usage."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "gCuSR8GkAgzl"
},
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Vertex AI Workbench Notebooks**, your environment already meets\n",
"all the requirements to run this notebook. You can skip this step."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "24743cf4a1e1"
},
"source": [
"**_NOTE_**: This notebook has been tested in the following environment:\n",
"\n",
"* Python version = 3.9"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "gCuSR8GkAgzl"
},
"source": [
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
"You need the following:\n",
"\n",
"* The Google Cloud SDK\n",
"\n",
"The Google Cloud guide to [Setting up a Python development\n",
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
"for meeting these requirements. The following steps provide a condensed set of\n",
"instructions:\n",
"\n",
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
"\n",
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
"\n",
"1. [Install\n",
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"1. To install Jupyter, run `pip3 install jupyter` on the\n",
"command-line in a terminal shell.\n",
"\n",
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"1. Open this notebook in the Jupyter Notebook Dashboard."
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -159,32 +202,51 @@
},
"outputs": [],
"source": [
"# Install the packages\n",
"! pip3 install --user --upgrade google-cloud-aiplatform"
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform {USER_FLAG} -q\n",
"# TODO: Add remaining package installs here. All packages should be on a single pip install to resolve dependencies"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "58707a750154"
"id": "hhq5zEbGg0XX"
},
"source": [
"### Colab only: Uncomment the following cell to restart the kernel."
"### Restart the kernel\n",
"\n",
"After you install the additional packages, you need to restart the notebook kernel so it can find the packages."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "f200f10a1da3"
"id": "EzrelQZ22IZj"
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs so that your environment can access the new packages\n",
"# import IPython\n",
"# Automatically restart kernel after installs\n",
"import os\n",
"\n",
"# app = IPython.Application.instance()\n",
"# app.kernel.do_shutdown(True)"
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
@@ -201,11 +263,16 @@
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"2. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"\n",
"3. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com). {TODO: Update the APIs needed for your tutorial. Edit the API names, and update the link to append the API IDs, separating each one with a comma. For example, container.googleapis.com,cloudbuild.googleapis.com}\n",
"1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com). {TODO: Update the APIs needed for your tutorial. Edit the API names, and update the link to append the API IDs, separating each one with a comma. For example, container.googleapis.com,cloudbuild.googleapis.com}\n",
"\n",
"4. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk)."
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
]
},
{
@@ -216,10 +283,7 @@
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, try the following:\n",
"* Run `gcloud config list`.\n",
"* Run `gcloud projects list`.\n",
"* See the support page: [Locate the project ID](https://support.google.com/googleapi/answer/7014113)"
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
@@ -230,10 +294,57 @@
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}\n",
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "riG_qUokg0XZ"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_gcloud_project_id"
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "23988890fef6"
},
"source": [
"#### Get your project number {TODO: Include these cells if the notebook uses a project number}\n",
"\n",
"# Set the project id\n",
"! gcloud config set project {PROJECT_ID}"
"Now that the project ID is set, you get your corresponding project number."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "2d6950574e1d"
},
"outputs": [],
"source": [
"shell_output = ! gcloud projects list --filter=\"PROJECT_ID:'{PROJECT_ID}'\" --format='value(PROJECT_NUMBER)'\n",
"PROJECT_NUMBER = shell_output[0]\n",
"print(\"Project Number:\", PROJECT_NUMBER)"
]
},
{
@@ -244,18 +355,63 @@
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable used by Vertex AI. Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. It is recommended that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
]
},
{
"cell_type": "code",
"execution_count": 1,
"execution_count": null,
"metadata": {
"id": "region"
},
"outputs": [],
"source": [
"REGION = \"us-central1\" # @param {type: \"string\"}"
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "06571eb4063b"
},
"source": [
"#### UUID\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial.\n",
"\n",
"{TODO: replace the `TIMESTAMP` with `UUID` in official notebooks}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "697568e92bd6"
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
]
},
{
@@ -266,68 +422,64 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"Depending on your Jupyter environment, you may have to manually authenticate. Follow the relevant instructions below."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "74ccc9e52986"
},
"source": [
"**1. Vertex AI Workbench**\n",
"* Do nothing as you are already authenticated."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "de775a3773ba"
},
"source": [
"**2. Local JupyterLab instance, uncomment and run:**"
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
"authenticated. \n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"\n",
"2. Click **Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
]
},
{
"cell_type": "code",
"execution_count": 2,
"execution_count": null,
"metadata": {
"id": "254614fa0c46"
"id": "PyQmSRbKA8r-"
},
"outputs": [],
"source": [
"# ! gcloud auth login"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ef21552ccea8"
},
"source": [
"**3. Colab, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": 3,
"metadata": {
"id": "603adbbf0532"
},
"outputs": [],
"source": [
"# from google.colab import auth\n",
"# auth.authenticate_user()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "f6b2ccc891ed"
},
"source": [
"**4. Service account or other**\n",
"* See how to grant Cloud Storage permissions to your service account at https://cloud.google.com/storage/docs/gsutil/commands/iam#ch-examples."
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS '[your-service-account-key-path]'"
]
},
{
@@ -338,9 +490,20 @@
"source": [
"### Create a Cloud Storage bucket\n",
"\n",
"Create a storage bucket to store intermediate artifacts such as datasets.\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"- *{Note to notebook author: For any user-provided strings that need to be unique (like bucket names or model ID's), append \"-unique\" to the end so proper testing can occur}*"
"\n",
"{TODO: Adjust wording in the first paragraph to fit your use case - explain how your tutorial uses the Cloud Storage bucket. The example below shows how Vertex AI uses the bucket for training.}\n",
"\n",
"When you submit a training job using the Vertex AI SDK, you upload a Python package\n",
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
"the code from this package. In this tutorial, Vertex AI also saves the\n",
"trained model that results from your job in the same bucket. Using this model artifact, you can then\n",
"create Vertex AI model and endpoint resources in order to serve\n",
"online predictions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. It must be unique across all\n",
"Cloud Storage buckets."
]
},
{
@@ -351,7 +514,21 @@
},
"outputs": [],
"source": [
"BUCKET_URI = \"gs://your-bucket-name-unique\" # @param {type:\"string\"}"
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cf221059d072"
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
@@ -377,7 +554,99 @@
{
"cell_type": "markdown",
"metadata": {
"id": "960505627ddf"
"id": "ucvCsknMCims"
},
"source": [
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "vhOb7YnwClBb"
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "set_service_account"
},
"source": [
"#### Service Account {TODO: Include these cells if the notebook specifies a service account}\n",
"\n",
"{TODO: What uses service account in the notebook; e.g., You use a service account to create Vertex AI Pipeline jobs.}. If you do not want to use your project's Compute Engine service account, set `SERVICE_ACCOUNT` to another service account ID."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_service_account"
},
"outputs": [],
"source": [
"SERVICE_ACCOUNT = \"[your-service-account]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_service_account"
},
"outputs": [],
"source": [
"if (\n",
" SERVICE_ACCOUNT == \"\"\n",
" or SERVICE_ACCOUNT is None\n",
" or SERVICE_ACCOUNT == \"[your-service-account]\"\n",
"):\n",
" # Get your service account from gcloud\n",
" if not IS_COLAB:\n",
" shell_output = !gcloud auth list 2>/dev/null\n",
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
"\n",
" else: # IS_COLAB:\n",
" shell_output = ! gcloud projects describe $PROJECT_ID\n",
" project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
" SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
"\n",
" print(\"Service Account:\", SERVICE_ACCOUNT)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "set_service_account:pipelines"
},
"source": [
"#### Set service account access for {TODO; e.g., Vertex AI Pipelines}\n",
"\n",
"Run the following commands to grant your service account access to {TODO; i.e., read and write pipeline artifacts} in the bucket that you created in the previous step. You only need to run this step once per service account."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_service_account:pipelines"
},
"outputs": [],
"source": [
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
"\n",
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "XoEqT2Y4DJmf"
},
"source": [
"### Import libraries"
@@ -387,11 +656,13 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "PyQmSRbKA8r-"
"id": "pRUOFELefqf1"
},
"outputs": [],
"source": [
"from google.cloud import aiplatform"
"import google.cloud.aiplatform as aiplatform\n",
"\n",
"# TODO: import remaining libraries; e.g., tensorflow"
]
},
{
@@ -434,21 +705,21 @@
},
{
"cell_type": "code",
"execution_count": 1,
"execution_count": null,
"metadata": {
"id": "sx_vKniMq9ZX"
},
"outputs": [],
"source": [
"import os\n",
"\n",
"# Delete endpoint resource\n",
"# e.g. `endpoint.delete()`\n",
"! gcloud ai endpoints delete $ENDPOINT_NAME --quiet --region $REGION\n",
"\n",
"# Delete model resource\n",
"# e.g. `model.delete()`\n",
"! gcloud ai models delete $MODEL_NAME --quiet\n",
"\n",
"# Delete Cloud Storage objects that were created\n",
"! gsutil -m rm -r $JOB_DIR\n",
"\n",
"delete_bucket = False\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
" ! gsutil -m rm -r $BUCKET_URI"
-2
View File
@@ -38,5 +38,3 @@
/model_evaluation/automl_tabular_regression_model_evaluation.ipynb @soheilazangeneh
/tabular_workflows/tabnet_on_vertex_pipelines.ipynb @sakagarwal
/tabular_workflows/wide_and_deep_on_vertex_pipelines.ipynb @sakagarwal
/model_evaluation/custom_tabular_classification_model_evaluation.ipynb @soheilazangeneh
/sdk/SDK_FBProphet_Forecasting_Online.ipynb @brianchunkang
@@ -8,7 +8,7 @@
},
"outputs": [],
"source": [
"# Copyright 2022 Google LLC\n",
"# Copyright 2021 Google LLC\n",
"#\n",
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
"# you may not use this file except in compliance with the License.\n",
@@ -66,17 +66,6 @@
"This tutorial demonstrates how to use the Vertex AI SDK to create image object detection models and do batch prediction using a Google Cloud [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users) model."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "dataset:salads,iod"
},
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and the corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese."
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -85,7 +74,7 @@
"source": [
"### Objective\n",
"\n",
"In this tutorial, you create an AutoML image object detection model from a Python script, and then do a batch prediction using the Vertex AI SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Cloud Console.\n",
"In this tutorial, you learn how to create an AutoML image object detection model from a Python script, and then do a batch prediction using the Vertex AI SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Cloud Console.\n",
"\n",
"The steps performed include:\n",
"\n",
@@ -101,6 +90,17 @@
"* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "dataset:salads,iod"
},
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and the corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese."
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -121,6 +121,39 @@
"to generate a cost estimate based on your projected usage."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "setup_local"
},
"source": [
"### Set up your local development environment\n",
"\n",
"If you are using Colab or Google Cloud Notebooks, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"\n",
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
"\n",
"- The Cloud Storage SDK\n",
"- Git\n",
"- Python 3\n",
"- virtualenv\n",
"- Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"The Cloud Storage guide to [Setting up a Python development environment](https://cloud.google.com/python/setup) and the [Jupyter installation guide](https://jupyter.org/install) provide detailed instructions for meeting these requirements. The following steps provide a condensed set of instructions:\n",
"\n",
"1. [Install and initialize the SDK](https://cloud.google.com/sdk/docs/).\n",
"\n",
"2. [Install Python 3](https://cloud.google.com/python/setup#installing_python).\n",
"\n",
"3. [Install virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv) and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"4. To install Jupyter, run `pip3 install jupyter` on the command-line in a terminal shell.\n",
"\n",
"5. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"6. Open this notebook in the Jupyter Notebook Dashboard.\n"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -129,7 +162,7 @@
"source": [
"## Installation\n",
"\n",
"Install the latest version of Cloud Storage, Bigquery and Vertex AI SDKs for Python."
"Install the latest version of Vertex AI SDK for Python."
]
},
{
@@ -140,10 +173,44 @@
},
"outputs": [],
"source": [
"# Install the packages.\n",
"! pip3 install --upgrade google-cloud-aiplatform \\\n",
" google-cloud-storage \\\n",
" tensorflow -q"
"import os\n",
"\n",
"# Google Cloud Notebook\n",
"if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" USER_FLAG = \"--user\"\n",
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_storage"
},
"source": [
"Install the latest GA version of *google-cloud-storage* library."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "d6Pa6Sybv5mK"
},
"outputs": [],
"source": [
"! pip3 install -U --upgrade tensorflow google-cloud-storage $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "9_zWlX10v5mL"
},
"source": [
"Install the latest version of *tensorflow* library."
]
},
{
@@ -152,7 +219,9 @@
"id": "restart"
},
"source": [
"### Colab only: Uncomment the following cell to restart the kernel"
"### Restart the kernel\n",
"\n",
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
]
},
{
@@ -163,20 +232,14 @@
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs so that your environment can access the new packages\n",
"# import IPython\n",
"import os\n",
"\n",
"# app = IPython.Application.instance()\n",
"# app.kernel.do_shutdown(True)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "013daf3de88e"
},
"source": [
"## Before you begin"
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
@@ -185,12 +248,28 @@
"id": "before_you_begin:nogpu"
},
"source": [
"### Set your project ID\n",
"## Before you begin\n",
"\n",
"**If you don't know your project ID**, try the following:\n",
"* Run `gcloud config list`.\n",
"* Run `gcloud projects list`.\n",
"* See the support page: [Locate the project ID](https://support.google.com/googleapi/answer/7014113)"
"### GPU runtime\n",
"\n",
"This tutorial does not require a GPU runtime.\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
"\n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
@@ -201,10 +280,56 @@
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}\n",
"import os\n",
"\n",
"# Set the project id\n",
"! gcloud config set project {PROJECT_ID}"
"PROJECT_ID = \"\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_project_id"
},
"outputs": [],
"source": [
"# Get your Google Cloud project ID from gcloud\n",
"if not os.getenv(\"IS_TESTING\"):\n",
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID: \", PROJECT_ID)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "W2F5WRyhv5mO"
},
"source": [
"Otherwise, set your project ID here."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "d7-MjQafv5mO"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_gcloud_project_id"
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
]
},
{
@@ -215,7 +340,16 @@
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable used by Vertex AI. Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
]
},
{
@@ -226,7 +360,41 @@
},
"outputs": [],
"source": [
"REGION = \"us-central1\" # @param {type: \"string\"}"
"REGION = \"[your-region]\" # @param {type:\"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "timestamp"
},
"source": [
"#### UUID\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "PzKW-zT_v5mR"
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
]
},
{
@@ -237,68 +405,53 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"Depending on your Jupyter environment, you may have to manually authenticate. Follow the relevant instructions below."
"**If you are using Google Cloud Notebooks**, your environment is already authenticated.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"**Click Create service account**.\n",
"\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
"cell_type": "markdown",
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "FvQeFm3Gv5mR"
},
"source": [
"**1. Vertex AI Workbench**\n",
"* Do nothing as you are already authenticated."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ad1138a125ea"
},
"source": [
"**2. Local JupyterLab instance, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "ce6043da7b33"
},
"outputs": [],
"source": [
"# ! gcloud auth login"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "0367eac06a10"
},
"source": [
"**3. Colab, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "21ad4dbb4a61"
},
"outputs": [],
"source": [
"# from google.colab import auth\n",
"# auth.authenticate_user()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "c13224697bfb"
},
"source": [
"**4. Service account or other**\n",
"* See how to grant Cloud Storage permissions to your service account at https://cloud.google.com/storage/docs/gsutil/commands/iam#ch-examples."
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
@@ -309,7 +462,11 @@
"source": [
"### Create a Cloud Storage bucket\n",
"\n",
"Create a storage bucket to store intermediate artifacts such as datasets."
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"When you initialize the Vertex AI SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
]
},
{
@@ -320,7 +477,20 @@
},
"outputs": [],
"source": [
"BUCKET_URI = \"gs://your-bucket-name-unique\" # @param {type:\"string\"}"
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_bucket"
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + UUID"
]
},
{
@@ -340,7 +510,27 @@
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "validate_bucket"
},
"source": [
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "N9JY-esPv5mU"
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
]
},
{
@@ -349,7 +539,10 @@
"id": "setup_vars"
},
"source": [
"### Import libraries"
"### Set up variables\n",
"\n",
"Next, set up some variables used throughout the tutorial.\n",
"### Import libraries and define constants"
]
},
{
@@ -382,7 +575,7 @@
},
"outputs": [],
"source": [
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI, location=REGION)"
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME, location=REGION)"
]
},
{
@@ -471,9 +664,8 @@
},
"outputs": [],
"source": [
"DISPLAY_NAME = \"salads_unique\"\n",
"dataset = aiplatform.ImageDataset.create(\n",
" display_name=DISPLAY_NAME,\n",
" display_name=\"Salads\" + \"_\" + UUID,\n",
" gcs_source=[IMPORT_FILE],\n",
" import_schema_uri=aiplatform.schema.dataset.ioformat.image.bounding_box,\n",
")\n",
@@ -521,7 +713,7 @@
"outputs": [],
"source": [
"job = aiplatform.AutoMLImageTrainingJob(\n",
" display_name=DISPLAY_NAME,\n",
" display_name=\"salads_\" + UUID,\n",
" prediction_type=\"object_detection\",\n",
" multi_label=False,\n",
" model_type=\"CLOUD\",\n",
@@ -564,7 +756,7 @@
"source": [
"model = job.run(\n",
" dataset=dataset,\n",
" model_display_name=DISPLAY_NAME,\n",
" model_display_name=\"salads_\" + UUID,\n",
" training_fraction_split=0.8,\n",
" validation_fraction_split=0.1,\n",
" test_fraction_split=0.1,\n",
@@ -594,8 +786,7 @@
"outputs": [],
"source": [
"# Get model resource ID\n",
"filter_name = f\"display_name={DISPLAY_NAME}\"\n",
"models = aiplatform.Model.list(filter=filter_name)\n",
"models = aiplatform.Model.list(filter=\"display_name=salads_\" + UUID)\n",
"\n",
"# Get a reference to the Model Service client\n",
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
@@ -714,7 +905,6 @@
"outputs": [],
"source": [
"import json\n",
"import os\n",
"\n",
"import tensorflow as tf\n",
"\n",
@@ -767,7 +957,7 @@
"outputs": [],
"source": [
"batch_predict_job = model.batch_predict(\n",
" job_display_name=DISPLAY_NAME,\n",
" job_display_name=\"salads_\" + UUID,\n",
" gcs_source=gcs_input_uri,\n",
" gcs_destination_prefix=BUCKET_URI,\n",
" machine_type=\"n1-standard-4\",\n",
@@ -74,14 +74,6 @@
"\n",
"In this tutorial, you learn how to create an AutoML tabular regression model and deploy it for batch prediction using the Vertex AI SDK for Python. You can alternatively create and deploy models using the `gcloud` command-line tool or batch using the Cloud Console.\n",
"\n",
"This tutorial uses the following Google Cloud ML services and resources:\n",
"\n",
"- Vertex AI Datasets (Tabular)\n",
"- Vertex AI Training (AutoML Tabular Training)\n",
"- Vertex AI Model Registry\n",
"- Vertex AI Endpoint\n",
"- Vertex AI Batch predictions\n",
"\n",
"The steps performed include:\n",
"\n",
"- Create a Vertex AI `Dataset` resource.\n",
@@ -115,11 +107,45 @@
"\n",
"* Vertex AI\n",
"* Cloud Storage\n",
"* BigQuery / BigQuery ML\n",
"\n",
"Learn about [Vertex AI\n",
"pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage\n",
"pricing](https://cloud.google.com/storage/pricing) and [BigQuery pricing](https://cloud.google.com/bigquery/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
"pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n",
"pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n",
"Calculator](https://cloud.google.com/products/calculator/)\n",
"to generate a cost estimate based on your projected usage."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "setup_local"
},
"source": [
"### Set up your local development environment\n",
"\n",
"If you are using Colab or Vertex AI Workbench, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"\n",
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
"\n",
"- The Cloud Storage SDK\n",
"- Git\n",
"- Python 3\n",
"- virtualenv\n",
"- Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"The Cloud Storage guide to [Setting up a Python development environment](https://cloud.google.com/python/setup) and the [Jupyter installation guide](https://jupyter.org/install) provide detailed instructions for meeting these requirements. The following steps provide a condensed set of instructions:\n",
"\n",
"1. [Install and initialize the SDK](https://cloud.google.com/sdk/docs/).\n",
"\n",
"2. [Install Python 3](https://cloud.google.com/python/setup#installing_python).\n",
"\n",
"3. [Install virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv) and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"4. To install Jupyter, run `pip3 install jupyter` on the command-line in a terminal shell.\n",
"\n",
"5. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"6. Open this notebook in the Jupyter Notebook Dashboard.\n"
]
},
{
@@ -130,22 +156,66 @@
"source": [
"## Installation\n",
"\n",
"Install the following packages required to execute this notebook."
"Install the latest version of the Vertex AI SDK for Python."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "870f1b093d9c"
"id": "install_aip:mbsdk"
},
"outputs": [],
"source": [
"# Install the packages\n",
"! pip3 install --upgrade --quiet google-cloud-aiplatform \\\n",
" 'google-cloud-bigquery[bqstorage,pandas]' \\\n",
" google-cloud-storage \n",
" "
"import os\n",
"\n",
"# Google Cloud Notebook\n",
"if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" USER_FLAG = \"--user\"\n",
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_storage"
},
"source": [
"Install the latest version of *google-cloud-storage*."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_storage"
},
"outputs": [],
"source": [
"! pip3 install -U google-cloud-storage $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "b2f8bf1a1c31"
},
"source": [
"Install the latest version of *google-cloud-bigquery*."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "fb18bb35a386"
},
"outputs": [],
"source": [
"! pip3 install -U \"google-cloud-bigquery[pandas]\" $USER_FLAG"
]
},
{
@@ -154,7 +224,9 @@
"id": "restart"
},
"source": [
"### Colab only: Uncomment the following cell to restart the kernel."
"### Restart the kernel\n",
"\n",
"After installing the packages, restart the notebook kernel."
]
},
{
@@ -165,11 +237,14 @@
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs so that your environment can access the new packages\n",
"# import IPython\n",
"import os\n",
"\n",
"# app = IPython.Application.instance()\n",
"# app.kernel.do_shutdown(True)"
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
@@ -179,12 +254,27 @@
},
"source": [
"## Before you begin\n",
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, try the following:\n",
"* Run `gcloud config list`.\n",
"* Run `gcloud projects list`.\n",
"* See the support page: [Locate the project ID](https://support.google.com/googleapi/answer/7014113)"
"### GPU runtime\n",
"\n",
"This tutorial does not require a GPU runtime.\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
"\n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"4. If you are running this notebook locally, install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
@@ -195,10 +285,33 @@
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}\n",
"\n",
"# Set the project id\n",
"! gcloud config set project {PROJECT_ID}"
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_project_id"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "set_gcloud_project_id"
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
]
},
{
@@ -209,7 +322,15 @@
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable used by Vertex AI. Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"You can also change the `REGION` variable, which is used for operations throughout the rest of this notebook. The following regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
]
},
{
@@ -223,6 +344,30 @@
"REGION = \"[your-region]\" # @param {type: \"string\"}"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "timestamp"
},
"source": [
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "timestamp"
},
"outputs": [],
"source": [
"from datetime import datetime\n",
"\n",
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -231,53 +376,53 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"Depending on your Jupyter environment, you may have to manually authenticate. Follow the relevant instructions below.\n",
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
"\n",
"**1. Vertex AI Workbench**\n",
"* Do nothing as you are already authenticated.\n",
"**2. Local JupyterLab instance, uncomment and run:**\n"
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"**Click Create service account**.\n",
"\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "457c78b08293"
"id": "gcp_authenticate"
},
"outputs": [],
"source": [
"# ! gcloud auth login"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "d3e571ce6c56"
},
"source": [
"**3. Colab, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "984a0526fb68"
},
"outputs": [],
"source": [
"# from google.colab import auth\n",
"# auth.authenticate_user()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "2c549a59cca4"
},
"source": [
"**4. Service account or other**\n",
"* See how to grant Cloud Storage permissions to your service account at https://cloud.google.com/storage/docs/gsutil/commands/iam#ch-examples."
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" ! gcloud auth login"
]
},
{
@@ -300,8 +445,7 @@
},
"outputs": [],
"source": [
"import google.cloud.aiplatform as aiplatform\n",
"from google.cloud import bigquery"
"import google.cloud.aiplatform as aiplatform"
]
},
{
@@ -378,6 +522,8 @@
},
"outputs": [],
"source": [
"from google.cloud import bigquery\n",
"\n",
"# Create client in default region\n",
"bq_client = bigquery.Client(\n",
" project=PROJECT_ID,\n",
@@ -394,13 +540,13 @@
"outputs": [],
"source": [
"# Create training dataset in default region\n",
"TRAINING_INPUT_DATASET_ID = \"gsod_training_unique\"\n",
"TRAINING_INPUT_DATASET_ID = f\"gsod_training_{TIMESTAMP}\"\n",
"bq_dataset = bigquery.Dataset(f\"{PROJECT_ID}.{TRAINING_INPUT_DATASET_ID}\")\n",
"bq_dataset = bq_client.create_dataset(bq_dataset)\n",
"print(f\"Created dataset {bq_client.project}.{bq_dataset.dataset_id}\")\n",
"\n",
"# Create test dataset in default region\n",
"PREDICTION_INPUT_DATASET_ID = \"gsod_prediction_unique\"\n",
"PREDICTION_INPUT_DATASET_ID = f\"gsod_prediction_{TIMESTAMP}\"\n",
"bq_dataset = bigquery.Dataset(f\"{PROJECT_ID}.{PREDICTION_INPUT_DATASET_ID}\")\n",
"bq_dataset = bq_client.create_dataset(bq_dataset)\n",
"print(f\"Created dataset {bq_client.project}.{bq_dataset.dataset_id}\")"
@@ -479,7 +625,7 @@
"outputs": [],
"source": [
"dataset = aiplatform.TabularDataset.create(\n",
" display_name=\"NOAA historical weather data_unique\",\n",
" display_name=\"NOAA historical weather data\" + \"_\" + TIMESTAMP,\n",
" bq_source=[f\"bq://{TRAINING_INPUT_TABLE_ID}\"],\n",
")\n",
"\n",
@@ -550,151 +696,7 @@
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
")\n",
"\n",
"print(training_job)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "create_automl_pipeline:tabular,lrg,transformations"
},
"outputs": [],
"source": [
"training_job = aiplatform.AutoMLTabularTrainingJob(\n",
" display_name=\"job_unique\",\n",
" display_name=\"gsod_\" + TIMESTAMP,\n",
" optimization_prediction_type=\"regression\",\n",
" optimization_objective=\"minimize-rmse\",\n",
" column_specs=COLUMN_SPECS,\n",
@@ -724,7 +726,7 @@
"\n",
"The `run` method when completed returns the `Model` resource.\n",
"\n",
"The execution of the training pipeline will take upto 3 hours."
"The execution of the training pipeline will take upto 20 minutes."
]
},
{
@@ -737,7 +739,7 @@
"source": [
"model = training_job.run(\n",
" dataset=dataset,\n",
" model_display_name=\"model_unique\",\n",
" model_display_name=\"gsod_\" + TIMESTAMP,\n",
" training_fraction_split=0.6,\n",
" validation_fraction_split=0.2,\n",
" test_fraction_split=0.2,\n",
@@ -803,7 +805,7 @@
"outputs": [],
"source": [
"# Create results dataset in default region\n",
"RESULTS_DATASET_ID = \"gsod_results_unique\"\n",
"RESULTS_DATASET_ID = f\"gsod_results_{TIMESTAMP}\"\n",
"bq_dataset = bigquery.Dataset(f\"{PROJECT_ID}.{RESULTS_DATASET_ID}\")\n",
"bq_dataset = bq_client.create_dataset(bq_dataset)\n",
"print(f\"Created dataset {bq_client.project}.{bq_dataset.dataset_id}\")"
@@ -827,9 +829,7 @@
"- `machine_type`: The type of machine to use for training.\n",
"- `accelerator_type`: The hardware accelerator type.\n",
"- `accelerator_count`: The number of accelerators to attach to a worker replica.\n",
"- `sync`: If set to True, the call will block while waiting for the asynchronous batch job to complete.\n",
"\n",
"Batch prediction job takes roughly 1 hour to finish."
"- `sync`: If set to True, the call will block while waiting for the asynchronous batch job to complete."
]
},
{
@@ -922,8 +922,17 @@
"\n",
"bq_client.delete_dataset(\n",
" f\"{PROJECT_ID}.{RESULTS_DATASET_ID}\", delete_contents=True, not_found_ok=True\n",
")\n",
"\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cleanup:mbsdk"
},
"outputs": [],
"source": [
"# Delete Vertex AI resources\n",
"dataset.delete()\n",
"model.delete()\n",
@@ -29,7 +29,7 @@
"id": "title"
},
"source": [
"# Vertex AI SDK: Training an AutoML text sentiment analysis model for online predictions\n",
"# Vertex SDK: AutoML training text sentiment analysis model for online prediction\n",
"\n",
"<table align=\"left\">\n",
" <td>\n",
@@ -45,7 +45,6 @@
" </td>\n",
" <td>\n",
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/automl/sdk_automl_text_sentiment_analysis_online.ipynb\">\n",
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
" Open in Vertex AI Workbench\n",
" </a>\n",
" </td>\n",
@@ -61,7 +60,8 @@
"source": [
"## Overview\n",
"\n",
"This tutorial demonstrates how to use the Vertex AI SDK to train and deploy an [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users) text sentiment analysis model and get online predictions from it."
"\n",
"This tutorial demonstrates how to use the Vertex AI SDK to create text sentiment analysis models and do online prediction using a Google Cloud [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users) model."
]
},
{
@@ -72,23 +72,16 @@
"source": [
"### Objective\n",
"\n",
"In this tutorial, you learn how to create an AutoML text sentiment analysis model and deploy it for online predictions from a Python script using the Vertex AI SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Cloud Console.\n",
"\n",
"This tutorial uses the following Google Cloud ML services and resources:\n",
"- Vertex AI Datasets\n",
"- Vertex AI Training (AutoML)\n",
"- Vertex AI Model Registry\n",
"- Vertex AI Endpoints\n",
"In this tutorial, you learn how to create an AutoML text sentiment analysis model and deploy for online prediction from a Python script using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Cloud Console.\n",
"\n",
"The steps performed include:\n",
"\n",
"- Create a `Vertex AI Dataset` resource.\n",
"- Create a training job for the AutoML model on the dataset.\n",
"- View the model evaluation metrics.\n",
"- Deploy the `Vertex AI Model` resource to a serving `Vertex AI Endpoint`.\n",
"- Make a prediction request to the deployed model.\n",
"- Undeploy the model from endpoint.\n",
"- Perform clean up process."
"- Create a Vertex `Dataset` resource.\n",
"- Create a training job for the model.\n",
"- View the model evaluation.\n",
"- Deploy the `Model` resource to a serving `Endpoint` resource.\n",
"- Make a prediction.\n",
"- Undeploy the `Model`."
]
},
{
@@ -99,7 +92,7 @@
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) that consists of tweets tagged with sentiment, the author's gender, and whether or not they mention any of the top 10 adverse events reported to the FDA. The version of the dataset you use in this tutorial is stored in a public Cloud Storage bucket. In this tutorial, you use the tweets data to build an AutoML text sentiment analysis model on Google Cloud platform."
"The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) that consists of tweets tagged with sentiment, the author's gender, and whether or not they mention any of the top 10 adverse events reported to the FDA. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. In this tutorial, you will use the tweets' data to build an AutoML-text-sentiment-analysis model on Google Cloud platform."
]
},
{
@@ -130,34 +123,29 @@
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Vertex AI Workbench Notebooks**, your environment already meets\n",
"all the requirements to run this notebook. You can skip this step.\n",
"If you are using Colab or Google Cloud Notebooks, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"\n",
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
"You need the following:\n",
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
"\n",
"* The Google Cloud SDK\n",
"- The Cloud Storage SDK\n",
"- Git\n",
"- Python 3\n",
"- virtualenv\n",
"- Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"The Google Cloud guide to [Setting up a Python development\n",
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
"for meeting these requirements. The following steps provide a condensed set of\n",
"instructions:\n",
"The Cloud Storage guide to [Setting up a Python development environment](https://cloud.google.com/python/setup) and the [Jupyter installation guide](https://jupyter.org/install) provide detailed instructions for meeting these requirements. The following steps provide a condensed set of instructions:\n",
"\n",
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
"1. [Install and initialize the SDK](https://cloud.google.com/sdk/docs/).\n",
"\n",
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
"2. [Install Python 3](https://cloud.google.com/python/setup#installing_python).\n",
"\n",
"1. [Install\n",
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"3. [Install virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv) and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"1. To install Jupyter, run `pip3 install jupyter` on the\n",
"command-line in a terminal shell.\n",
"4. To install Jupyter, run `pip3 install jupyter` on the command-line in a terminal shell.\n",
"\n",
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"5. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"1. Open this notebook in the Jupyter Notebook Dashboard."
"6. Open this notebook in the Jupyter Notebook Dashboard.\n"
]
},
{
@@ -168,7 +156,7 @@
"source": [
"## Installation\n",
"\n",
"Install the latest version of Vertex AI SDK for Python."
"Install the latest version of Vertex SDK for Python."
]
},
{
@@ -181,19 +169,35 @@
"source": [
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
"# Google Cloud Notebook\n",
"if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" USER_FLAG = \"--user\"\n",
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q\n",
"! pip3 install -U google-cloud-storage $USER_FLAG -q"
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_storage"
},
"source": [
"Install the latest GA version of *google-cloud-storage* library as well.\n",
"\n",
"**Note**: You may encounter a PIP dependency error during the installation of the Google Cloud Storage package. This can be ignored as it will not affect the proper running of this script."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_storage"
},
"outputs": [],
"source": [
"! pip3 install -U google-cloud-storage $USER_FLAG"
]
},
{
@@ -204,7 +208,7 @@
"source": [
"### Restart the kernel\n",
"\n",
"After you install the additional packages, you need to restart the notebook kernel so it can find the packages."
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages. The following cell will restart the kernel."
]
},
{
@@ -215,7 +219,6 @@
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs\n",
"import os\n",
"\n",
"if not os.getenv(\"IS_TESTING\"):\n",
@@ -234,33 +237,26 @@
"source": [
"## Before you begin\n",
"\n",
"### GPU runtime\n",
"\n",
"This tutorial does not require a GPU runtime.\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
"\n",
"1. [Enable the Vertex AI, Compute Engine, and Cloud Storage APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "af794e75b7e3"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
@@ -340,9 +336,9 @@
"id": "timestamp"
},
"source": [
"#### UUID\n",
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
]
},
{
@@ -353,16 +349,9 @@
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
@@ -373,31 +362,23 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
"authenticated. \n",
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"2. Click **Create service account**.\n",
"**Click Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
@@ -416,11 +397,8 @@
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
@@ -430,7 +408,7 @@
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS '[your-service-account-key-path]'"
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
@@ -443,9 +421,9 @@
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"When you initialize your Vertex AI SDK, you provide a Cloud Storage bucket to the SDK to serve as a staging bucket for the session. \n",
"When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets."
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
]
},
{
@@ -456,8 +434,7 @@
},
"outputs": [],
"source": [
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
]
},
{
@@ -468,9 +445,8 @@
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
]
},
{
@@ -490,7 +466,7 @@
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_NAME"
]
},
{
@@ -510,7 +486,7 @@
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
"! gsutil ls -al $BUCKET_NAME"
]
},
{
@@ -519,7 +495,10 @@
"id": "setup_vars"
},
"source": [
"### Import libraries"
"### Set up variables\n",
"\n",
"Next, set up some variables used throughout the tutorial.\n",
"### Import libraries and define constants"
]
},
{
@@ -539,9 +518,9 @@
"id": "init_aip:mbsdk"
},
"source": [
"### Initialize Vertex AI SDK for Python\n",
"## Initialize Vertex SDK for Python\n",
"\n",
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
]
},
{
@@ -552,7 +531,18 @@
},
"outputs": [],
"source": [
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
"aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "tutorial_start:automl"
},
"source": [
"# Tutorial\n",
"\n",
"Now you are ready to start creating your own AutoML text sentiment analysis model."
]
},
{
@@ -561,9 +551,9 @@
"id": "import_file:u_dataset,csv"
},
"source": [
"### Define the constants\n",
"#### Location of Cloud Storage training data.\n",
"\n",
"Set the constants that you use in this tutorial."
"Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage."
]
},
{
@@ -574,9 +564,7 @@
},
"outputs": [],
"source": [
"# Set the location of the CSV index file in Cloud Storage.\n",
"IMPORT_FILE = \"gs://cloud-samples-data/language/claritin.csv\"\n",
"# Set the max. sentiment score\n",
"SENTIMENT_MAX = 4"
]
},
@@ -586,11 +574,11 @@
"id": "quick_peek:csv"
},
"source": [
"## Take a quick peek at your data\n",
"#### Quick peek at your data\n",
"\n",
"This tutorial uses a version of the `Crowdflower Claritin-Twitter` dataset which is stored in a public Cloud Storage bucket, using a CSV index file.\n",
"This tutorial uses a version of the Crowdflower Claritin-Twitter dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n",
"\n",
"Start by taking a quick peek at the data. Further, count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then print the first few rows."
"Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows."
]
},
{
@@ -616,12 +604,12 @@
"id": "create_dataset:text,tst"
},
"source": [
"## Create the Dataset\n",
"### Create the Dataset\n",
"\n",
"Now, create a `Vertex AI Dataset` resource using the `create` method of the `TextDataset` class, which takes the following parameters:\n",
"Next, create the `Dataset` resource using the `create` method for the `TextDataset` class, which takes the following parameters:\n",
"\n",
"- `display_name`: The human readable name for the dataset resource.\n",
"- `gcs_source`: A list of one or more dataset index files to import the data items into the dataset resource.\n",
"- `display_name`: The human readable name for the `Dataset` resource.\n",
"- `gcs_source`: A list of one or more dataset index files to import the data items into the `Dataset` resource.\n",
"- `import_schema_uri`: The data labeling schema for the data items.\n",
"\n",
"This operation may take several minutes."
@@ -636,7 +624,7 @@
"outputs": [],
"source": [
"dataset = aiplatform.TextDataset.create(\n",
" display_name=\"Crowdflower Claritin-Twitter\" + \"_\" + UUID,\n",
" display_name=\"Crowdflower Claritin-Twitter\" + \"_\" + TIMESTAMP,\n",
" gcs_source=[IMPORT_FILE],\n",
" import_schema_uri=aiplatform.schema.dataset.ioformat.text.sentiment,\n",
")\n",
@@ -650,18 +638,15 @@
"id": "create_automl_pipeline:text,tst"
},
"source": [
"## Create and run training job\n",
"### Create and run training pipeline\n",
"\n",
"In this section, to train an AutoML model, you perform these steps:\n",
"To train an AutoML model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n",
"\n",
"1) create a training job.\n",
"2) run the job.\n",
"#### Create training pipeline\n",
"\n",
"### Create a training job\n",
"An AutoML training pipeline is created with the `AutoMLTextTrainingJob` class, with the following parameters:\n",
"\n",
"An AutoML training job is created with the `AutoMLTextTrainingJob` class, with the following parameters:\n",
"\n",
"- `display_name`: The human readable name for the training job resource.\n",
"- `display_name`: The human readable name for the `TrainingJob` resource.\n",
"- `prediction_type`: The type task to train the model for.\n",
" - `classification`: A text classification model.\n",
" - `sentiment`: A text sentiment analysis model.\n",
@@ -679,7 +664,7 @@
"outputs": [],
"source": [
"job = aiplatform.AutoMLTextTrainingJob(\n",
" display_name=\"claritin_\" + UUID,\n",
" display_name=\"claritin_\" + TIMESTAMP,\n",
" prediction_type=\"sentiment\",\n",
" sentiment_max=SENTIMENT_MAX,\n",
")\n",
@@ -693,7 +678,7 @@
"id": "run_automl_pipeline:text"
},
"source": [
"### Run the training job\n",
"#### Run the training pipeline\n",
"\n",
"Next, you run the training job by invoking the method `run`, with the following parameters:\n",
"\n",
@@ -705,7 +690,7 @@
"\n",
"The `run` method when completed returns the `Model` resource.\n",
"\n",
"The execution of the training pipeline take upto 180 minutes."
"The execution of the training pipeline will take upto 180 minutes."
]
},
{
@@ -718,7 +703,7 @@
"source": [
"model = job.run(\n",
" dataset=dataset,\n",
" model_display_name=\"claritin_\" + UUID,\n",
" model_display_name=\"claritin_\" + TIMESTAMP,\n",
" training_fraction_split=0.8,\n",
" validation_fraction_split=0.1,\n",
" test_fraction_split=0.1,\n",
@@ -732,10 +717,9 @@
},
"source": [
"## Review model evaluation scores\n",
"After your model has finished training, you can review the evaluation scores for it.\n",
"\n",
"Once your model training has finished, you can review the evaluation scores.\n",
"\n",
"Firstly, you need to get a reference to the newly created model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project and filter."
"First, you need to get a reference to the new model. As with datasets, you can either use the reference to the model variable you created when you deployed the model or you can list all of the models in your project."
]
},
{
@@ -747,7 +731,7 @@
"outputs": [],
"source": [
"# Get model resource ID\n",
"models = aiplatform.Model.list(filter=\"display_name=claritin_\" + UUID)\n",
"models = aiplatform.Model.list(filter=\"display_name=claritin_\" + TIMESTAMP)\n",
"\n",
"# Get a reference to the Model Service client\n",
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
@@ -770,9 +754,7 @@
"source": [
"## Deploy the model\n",
"\n",
"Next, deploy your model to serve online predictions. To deploy the model, you invoke the `deploy` method of the model resource which in turn returns you the deployed endpoint.\n",
"\n",
"**Note:** Normally, an endpoint is created beforehand and is given as a reference while model deployment. By default, `deploy()` method creates an endpoint when an endpoint reference is not given."
"Next, deploy your model for online prediction. To deploy the model, you invoke the `deploy` method."
]
},
{
@@ -792,9 +774,9 @@
"id": "make_prediction"
},
"source": [
"## Send online prediction requests\n",
"## Send a online prediction request\n",
"\n",
"In this step, you prepare some test instances from the dataset and send an online prediction request to your deployed model."
"Send a online prediction to your deployed model."
]
},
{
@@ -803,9 +785,9 @@
"id": "get_test_item"
},
"source": [
"### Create test instances\n",
"### Get test item\n",
"\n",
"You use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model. It is just to demonstrate how to make a prediction."
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction."
]
},
{
@@ -831,21 +813,21 @@
"id": "predict_request:mbsdk,tst"
},
"source": [
"### Make the prediction request\n",
"### Make the prediction\n",
"\n",
"Now that your model is deployed to an endpoint, you can send online prediction requests to the endpoint resource.\n",
"Now that your `Model` resource is deployed to an `Endpoint` resource, you can do online predictions by sending prediction requests to the `Endpoint` resource.\n",
"\n",
"#### Request format\n",
"#### Request\n",
"\n",
"The format of each instance should be in JSON as below:\n",
"The format of each instance is:\n",
"\n",
" { 'content': text_string }\n",
"\n",
"Since the `predict()` method can take multiple instances, send your request as a list of one test instance.\n",
"Since the predict() method can take multiple items (instances), send your single test item as a list of one test item.\n",
"\n",
"#### Response\n",
"\n",
"The response from the `predict()` call is a Python dictionary with the following entries:\n",
"The response from the predict() call is a Python dictionary with the following entries:\n",
"\n",
"- `ids`: The internal assigned unique identifiers for each prediction request.\n",
"- `sentiment`: The sentiment value.\n",
@@ -874,7 +856,7 @@
"source": [
"## Undeploy the model\n",
"\n",
"After you explore the predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
"When you are done doing predictions, you undeploy the model from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model."
]
},
{
@@ -901,11 +883,11 @@
"\n",
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
"\n",
"- Vertex AI Dataset\n",
"- Vertex AI Model\n",
"- Vertex AI Endpoint\n",
"- Dataset\n",
"- Model\n",
"- Endpoint\n",
"- AutoML Training Job\n",
"- Cloud Storage Bucket (set `delete_bucket` to **True** to delete the bucket)"
"- Cloud Storage Bucket"
]
},
{
@@ -916,8 +898,6 @@
},
"outputs": [],
"source": [
"delete_bucket = False\n",
"\n",
"# Delete the dataset using the Vertex dataset object\n",
"dataset.delete()\n",
"\n",
@@ -931,8 +911,8 @@
"job.delete()\n",
"\n",
"# Delete the Cloud storage bucket\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
" ! gsutil rm -r $BUCKET_URI"
"if os.getenv(\"IS_TESTING\"):\n",
" ! gsutil rm -r $BUCKET_NAME"
]
}
],
@@ -18,7 +18,7 @@
"#\n",
"# Unless required by applicable law or agreed to in writing, software\n",
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
"# See the License for the specific language governing permissions and\n",
"# limitations under the License."
]
@@ -73,11 +73,7 @@
"source": [
"### Objective\n",
"\n",
"In this tutorial, you learn how to create an AutoML video object tracking model from a Python script, and then do a batch prediction using the Vertex AI SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Cloud Console.\n",
"\n",
"This tutorial uses the following Google Cloud ML services and resources:\n",
"\n",
"- Vertex AI\n",
"In this tutorial, you learn how to create an AutoML video object tracking model from a Python script, and then do a batch prediction using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Cloud Console.\n",
"\n",
"The steps performed include:\n",
"\n",
@@ -101,7 +97,7 @@
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the [Traffic](https://storage.googleapis.com/automl-video-demo-data/traffic_videos/traffic_videos_labels.csv) dataset. The version of the dataset you use in this tutorial is stored in a public Cloud Storage bucket."
"The dataset used for this tutorial is the [Traffic](https://storage.googleapis.com/automl-video-demo-data/traffic_videos/traffic_videos_labels.csv) dataset. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket."
]
},
{
@@ -132,11 +128,12 @@
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Vertex AI Workbench Notebooks**, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"If you are using Colab or Google Cloud Notebooks, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"\n",
"**Otherwise**, make sure your environment meets this notebook's requirements. You need the following:\n",
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
"\n",
"- The Cloud Storage SDK\n",
"- Git\n",
"- Python 3\n",
"- virtualenv\n",
"- Jupyter notebook running in a virtual environment with Python 3\n",
@@ -164,7 +161,7 @@
"source": [
"## Installation\n",
"\n",
"Install the following packages required to execute this notebook."
"Install the latest version of Vertex SDK for Python."
]
},
{
@@ -177,19 +174,33 @@
"source": [
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
"# Google Cloud Notebook\n",
"if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" USER_FLAG = \"--user\"\n",
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform google-cloud-storage $USER_FLAG -q"
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_storage"
},
"source": [
"Install the latest GA version of *google-cloud-storage* library as well."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_storage"
},
"outputs": [],
"source": [
"! pip3 install -U google-cloud-storage $USER_FLAG"
]
},
{
@@ -200,9 +211,7 @@
"source": [
"### Restart the kernel\n",
"\n",
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages.\n",
"\n",
"**Note: You may get a message saying \"Your session crashed for an unknown reason.\", this is expected. Once this cell has finished running, continue on. You do not need to re-run any of the cells above.**"
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
]
},
{
@@ -245,43 +254,30 @@
"\n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"4. If you are running this notebook locally, you need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"6. (optional) You may also specify a service account to use to run Vertex AI Pipelines in the project.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "5aee4379e8e5"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "dcdfccf50581"
"id": "set_project_id"
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
"PROJECT_ID = \"\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "5bf9979b96ff"
"id": "autoset_project_id"
},
"outputs": [],
"source": [
@@ -296,7 +292,7 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "09021c90b34c"
"id": "set_gcloud_project_id"
},
"outputs": [],
"source": [
@@ -342,9 +338,9 @@
"id": "timestamp"
},
"source": [
"#### UUID\n",
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
]
},
{
@@ -355,16 +351,9 @@
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
@@ -375,23 +364,23 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated.\n",
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"- In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"- **Click Create service account**.\n",
"**Click Create service account**.\n",
"\n",
"- In the **Service account name** field, enter a name, and click **Create**.\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"- In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"- Click Create. A JSON file that contains your key downloads to your local environment.\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"- Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
@@ -410,11 +399,8 @@
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
@@ -424,7 +410,7 @@
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS '[your-service-account-key-path]'"
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
@@ -437,7 +423,7 @@
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"When you initialize the Vertex AI SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
"When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
]
@@ -446,31 +432,29 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "f78cf4290843"
"id": "bucket"
},
"outputs": [],
"source": [
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"BUCKET_URI = \"\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "219a24ea078b"
"id": "autoset_bucket"
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "a8a62bec0259"
"id": "create_bucket"
},
"source": [
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
@@ -480,7 +464,7 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "91c46850b49b"
"id": "create_bucket"
},
"outputs": [],
"source": [
@@ -490,7 +474,7 @@
{
"cell_type": "markdown",
"metadata": {
"id": "4e69d430073b"
"id": "validate_bucket"
},
"source": [
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
@@ -500,7 +484,7 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "835eaacd691f"
"id": "validate_bucket"
},
"outputs": [],
"source": [
@@ -513,6 +497,9 @@
"id": "setup_vars"
},
"source": [
"### Set up variables\n",
"\n",
"Next, set up some variables used throughout the tutorial.\n",
"### Import libraries and define constants"
]
},
@@ -537,9 +524,9 @@
"id": "init_aip:mbsdk"
},
"source": [
"## Initialize Vertex AI SDK for Python\n",
"## Initialize Vertex SDK for Python\n",
"\n",
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
]
},
{
@@ -640,7 +627,7 @@
"outputs": [],
"source": [
"dataset = aiplatform.VideoDataset.create(\n",
" display_name=\"Traffic\" + \"_\" + UUID,\n",
" display_name=\"Traffic\" + \"_\" + TIMESTAMP,\n",
" gcs_source=[IMPORT_FILE],\n",
" import_schema_uri=aiplatform.schema.dataset.ioformat.video.object_tracking,\n",
")\n",
@@ -678,7 +665,7 @@
"outputs": [],
"source": [
"job = aiplatform.AutoMLVideoTrainingJob(\n",
" display_name=\"traffic_\" + UUID,\n",
" display_name=\"traffic_\" + TIMESTAMP,\n",
" prediction_type=\"object_tracking\",\n",
")\n",
"\n",
@@ -702,7 +689,7 @@
"\n",
"The `run` method when completed returns the `Model` resource.\n",
"\n",
"The execution of the training pipeline will take upto 4 hours."
"The execution of the training pipeline will take upto 5 hours."
]
},
{
@@ -715,7 +702,7 @@
"source": [
"model = job.run(\n",
" dataset=dataset,\n",
" model_display_name=\"traffic_\" + UUID,\n",
" model_display_name=\"traffic_\" + TIMESTAMP,\n",
" training_fraction_split=0.8,\n",
" test_fraction_split=0.2,\n",
")"
@@ -741,28 +728,22 @@
},
"outputs": [],
"source": [
"model_evaluations = model.list_model_evaluations()\n",
"# Get model resource ID\n",
"models = aiplatform.Model.list(filter=\"display_name=traffic_\" + TIMESTAMP)\n",
"\n",
"# Get a reference to the Model Service client\n",
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
"model_service_client = aiplatform.gapic.ModelServiceClient(\n",
" client_options=client_options\n",
")\n",
"\n",
"model_evaluations = model_service_client.list_model_evaluations(\n",
" parent=models[0].resource_name\n",
")\n",
"model_evaluation = list(model_evaluations)[0]\n",
"print(model_evaluation)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "b3377983702b"
},
"outputs": [],
"source": [
"# Print the evaluation metrics\n",
"for evaluation in model_evaluations:\n",
" evaluation = evaluation.to_dict()\n",
" print(\"Model's evaluation metrics from Training:\\n\")\n",
" metrics = evaluation[\"metrics\"]\n",
" for metric in metrics.keys():\n",
" print(f\"metric: {metric}, value: {metrics[metric]}\\n\")"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -893,7 +874,7 @@
"outputs": [],
"source": [
"batch_predict_job = model.batch_predict(\n",
" job_display_name=\"traffic_\" + UUID,\n",
" job_display_name=\"traffic_\" + TIMESTAMP,\n",
" gcs_source=gcs_input_uri,\n",
" gcs_destination_prefix=BUCKET_URI,\n",
" sync=False,\n",
@@ -985,9 +966,13 @@
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
"\n",
"- Dataset\n",
"- Pipeline\n",
"- Model\n",
"- Endpoint\n",
"- AutoML Training Job\n",
"- Batch Job\n",
"- Custom Job\n",
"- Hyperparameter Tuning Job\n",
"- Cloud Storage Bucket"
]
},
@@ -1011,8 +996,7 @@
"# Delete the batch prediction job using the Vertex batch prediction object\n",
"batch_predict_job.delete()\n",
"\n",
"delete_bucket = False\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
"if os.getenv(\"IS_TESTING\"):\n",
" ! gsutil -m rm -r $BUCKET_URI"
]
}
@@ -282,17 +282,6 @@
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "3c8049930470"
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -301,11 +290,15 @@
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
"PROJECT_ID = \"[YOUR-PROJECT-ID]\"\n",
"\n",
"# Get your Google Cloud project ID from gcloud\n",
"import os\n",
"\n",
"if not os.getenv(\"IS_TESTING\"):\n",
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
" print(\"Project ID: \", PROJECT_ID)"
]
},
{
@@ -325,7 +318,8 @@
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -123,6 +123,48 @@
"to generate a cost estimate based on your projected usage."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "a5cb73702a9b"
},
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Workbench AI Notebooks**, your environment already meets\n",
"all the requirements to run this notebook. You can skip this step.\n",
"\n",
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
"You need the following:\n",
"\n",
"* The Google Cloud SDK\n",
"* Git\n",
"* Python 3\n",
"* virtualenv\n",
"* Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"The Google Cloud guide to [Setting up a Python development\n",
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
"for meeting these requirements. The following steps provide a condensed set of\n",
"instructions:\n",
"\n",
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
"\n",
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
"\n",
"1. [Install\n",
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"1. To install Jupyter, run `pip install jupyter` on the\n",
"command-line in a terminal shell.\n",
"\n",
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"1. Open this notebook in the Jupyter Notebook Dashboard."
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -142,194 +184,306 @@
},
"outputs": [],
"source": [
"# Install the packages\n",
"! pip3 install --upgrade google-cloud-aiplatform \\\n",
" google-cloud-storage \\\n",
" pillow \\\n",
" numpy "
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install {USER_FLAG} --upgrade google-cloud-aiplatform -q\n",
"! pip3 install {USER_FLAG} --upgrade google-cloud-storage -q\n",
"! pip3 install {USER_FLAG} --upgrade pillow -q\n",
"! pip3 install {USER_FLAG} --upgrade numpy -q"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "d98bc9fdd80d"
"id": "restart"
},
"source": [
"### Colab only: Uncomment the following cell to restart the kernel."
"### Restart the kernel\n",
"\n",
"Once you've installed everything, you need to restart the notebook kernel so it can find the packages."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "0dbf29389c65"
"id": "bzPxhxS5lugp"
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs so that your environment can access the new packages\n",
"# import IPython\n",
"import os\n",
"\n",
"# app = IPython.Application.instance()\n",
"# app.kernel.do_shutdown(True)"
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "013daf3de88e"
"id": "before_you_begin"
},
"source": [
"## Before you begin"
"## Before you begin\n",
"\n",
"### Select a GPU runtime\n",
"\n",
"**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"2. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"\n",
"3. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n",
"\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "8bc8a29f9001"
"id": "project_id"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, try the following:\n",
"* Run `gcloud config list`.\n",
"* Run `gcloud projects list`.\n",
"* See the support page: [Locate the project ID](https://support.google.com/googleapi/answer/7014113)"
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "e61aaa036444"
"id": "autoset_project_id"
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}\n",
"\n",
"# Set the project id\n",
"! gcloud config set project {PROJECT_ID}"
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "USd_pUT0lugr"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "09021c90b34c"
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "c4a624c8099d"
"id": "88dd74c4c84e"
},
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable used by Vertex AI. Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "f83bd6013894"
"id": "5c615e53149f"
},
"outputs": [],
"source": [
"REGION = \"us-central1\" # @param {type: \"string\"}"
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "08bfd1eb44ef"
"id": "timestamp"
},
"source": [
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "c-pX32xalugs"
},
"outputs": [],
"source": [
"from datetime import datetime\n",
"\n",
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "gcp_authenticate"
},
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"Depending on your Jupyter environment, you may have to manually authenticate. Follow the relevant instructions below."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "af349043f23b"
},
"source": [
"**1. Vertex AI Workbench**\n",
"* Do nothing as you are already authenticated."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ad1138a125ea"
},
"source": [
"**2. Local JupyterLab instance, uncomment and run:**"
"**If you are using Workbench AI Notebooks**, your environment is already\n",
"authenticated. Skip this step.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"\n",
"2. Click **Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "ce6043da7b33"
"id": "vF60K5v1lugs"
},
"outputs": [],
"source": [
"# ! gcloud auth login"
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "0367eac06a10"
},
"source": [
"**3. Colab, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "21ad4dbb4a61"
},
"outputs": [],
"source": [
"# from google.colab import auth\n",
"# auth.authenticate_user()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "c13224697bfb"
},
"source": [
"**4. Service account or other**\n",
"* See how to grant Cloud Storage permissions to your service account at https://cloud.google.com/storage/docs/gsutil/commands/iam#ch-examples."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ddbea904fbe5"
"id": "bucket:custom"
},
"source": [
"### Create a Cloud Storage bucket\n",
"\n",
"Create a storage bucket to store intermediate artifacts such as datasets."
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"When you submit a training job using the Cloud SDK, you upload a Python package\n",
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
"the code from this package. In this tutorial, Vertex AI also saves the\n",
"trained model that results from your job in the same bucket. Using this model artifact, you can then create Vertex AI model resources.\n",
"\n",
"Set the name of your Cloud Storage bucket below. It must be unique across all\n",
"Cloud Storage buckets."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "751138cf3bd5"
"id": "bucket"
},
"outputs": [],
"source": [
"BUCKET_URI = \"gs://your-bucket-name-unique\" # @param {type:\"string\"}"
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_bucket"
},
"outputs": [],
"source": [
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "58cb4f5895f0"
"id": "create_bucket"
},
"source": [
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
@@ -339,11 +493,31 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "5e1288505682"
"id": "Oz8J0vmSlugt"
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "validate_bucket"
},
"source": [
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "oadE10x2lugu"
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
]
},
{
@@ -364,6 +538,7 @@
"outputs": [],
"source": [
"import os\n",
"import sys\n",
"\n",
"from google.cloud import aiplatform"
]
@@ -390,17 +565,54 @@
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_URI)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "accelerators:training,prediction"
},
"source": [
"#### Set hardware accelerators\n",
"\n",
"You can set hardware accelerators for training and prediction.\n",
"\n",
"Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n",
"\n",
" (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
"\n",
"\n",
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
"\n",
"Learn more about [hardware accelerator support for your region](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators).\n",
"\n",
"*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3. This is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "xd5PLXDTlugv"
},
"outputs": [],
"source": [
"TRAIN_GPU, TRAIN_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)\n",
"\n",
"DEPLOY_GPU, DEPLOY_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "container:training,prediction"
},
"source": [
"### Set pre-built containers\n",
"#### Set pre-built containers\n",
"\n",
"Vertex AI provides pre-built containers to run training and prediction.\n",
"Set the pre-built Docker container image for training and prediction.\n",
"\n",
"For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers) and [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)"
"For the latest list, see [Pre-built containers for training](https://cloud.google.com/ai-platform-unified/docs/training/pre-built-containers).\n",
"\n",
"For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/ai-platform-unified/docs/predictions/pre-built-containers)."
]
},
{
@@ -411,11 +623,64 @@
},
"outputs": [],
"source": [
"TRAIN_VERSION = \"tf-cpu.2-9\"\n",
"DEPLOY_VERSION = \"tf2-cpu.2-9\"\n",
"TRAIN_VERSION = \"tf-gpu.2-1\"\n",
"DEPLOY_VERSION = \"tf2-gpu.2-1\"\n",
"\n",
"TRAIN_IMAGE = \"us-docker.pkg.dev/vertex-ai/training/{}:latest\".format(TRAIN_VERSION)\n",
"DEPLOY_IMAGE = \"us-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(DEPLOY_VERSION)"
"TRAIN_IMAGE = \"{}-docker.pkg.dev/vertex-ai/training/{}:latest\".format(\n",
" REGION.split(\"-\")[0], TRAIN_VERSION\n",
")\n",
"DEPLOY_IMAGE = \"{}-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(\n",
" REGION.split(\"-\")[0], DEPLOY_VERSION\n",
")\n",
"\n",
"print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n",
"print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "machine:training,prediction"
},
"source": [
"#### Set machine types\n",
"\n",
"Next, set the machine types to use for training and prediction.\n",
"\n",
"- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure your compute resources for training and prediction.\n",
" - `machine type`\n",
" - `n1-standard`: 3.75GB of memory per vCPU\n",
" - `n1-highmem`: 6.5GB of memory per vCPU\n",
" - `n1-highcpu`: 0.9 GB of memory per vCPU\n",
" - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n",
"\n",
"*Note: The following is not supported for training:*\n",
"\n",
" - `standard`: 2 vCPUs\n",
" - `highcpu`: 2, 4 and 8 vCPUs\n",
"\n",
"*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "YAXwbqKKlugv"
},
"outputs": [],
"source": [
"MACHINE_TYPE = \"n1-standard\"\n",
"\n",
"VCPU = \"4\"\n",
"TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n",
"print(\"Train machine type\", TRAIN_COMPUTE)\n",
"\n",
"MACHINE_TYPE = \"n1-standard\"\n",
"\n",
"VCPU = \"4\"\n",
"DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n",
"print(\"Deploy machine type\", DEPLOY_COMPUTE)"
]
},
{
@@ -470,11 +735,13 @@
},
"outputs": [],
"source": [
"JOB_NAME = \"custom_job_unique\"\n",
"JOB_NAME = \"custom_job_\" + TIMESTAMP\n",
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, JOB_NAME)\n",
"\n",
"\n",
"TRAIN_STRATEGY = \"single\"\n",
"if not TRAIN_NGPU or TRAIN_NGPU < 2:\n",
" TRAIN_STRATEGY = \"single\"\n",
"else:\n",
" TRAIN_STRATEGY = \"mirror\"\n",
"\n",
"EPOCHS = 20\n",
"STEPS = 100\n",
@@ -656,15 +923,26 @@
" model_serving_container_image_uri=DEPLOY_IMAGE,\n",
")\n",
"\n",
"MODEL_DISPLAY_NAME = \"model_unique\"\n",
"MODEL_DISPLAY_NAME = \"cifar10-\" + TIMESTAMP\n",
"\n",
"# Start the training\n",
"\n",
"model = job.run(\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" args=CMDARGS,\n",
" replica_count=1,\n",
")"
"if TRAIN_GPU:\n",
" model = job.run(\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" args=CMDARGS,\n",
" replica_count=1,\n",
" machine_type=TRAIN_COMPUTE,\n",
" accelerator_type=TRAIN_GPU.name,\n",
" accelerator_count=TRAIN_NGPU,\n",
" )\n",
"else:\n",
" model = job.run(\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" args=CMDARGS,\n",
" replica_count=1,\n",
" machine_type=TRAIN_COMPUTE,\n",
" accelerator_count=0,\n",
" )"
]
},
{
@@ -833,7 +1111,7 @@
"MAX_NODES = 1\n",
"\n",
"# The name of the job\n",
"BATCH_PREDICTION_JOB_NAME = \"cifar10_batch_prediction_unique\"\n",
"BATCH_PREDICTION_JOB_NAME = \"cifar10_batch-\" + TIMESTAMP\n",
"\n",
"# Folder in the bucket to write results to\n",
"DESTINATION_FOLDER = \"batch_prediction_results\"\n",
@@ -849,9 +1127,11 @@
" gcs_source=BATCH_PREDICTION_GCS_SOURCE,\n",
" gcs_destination_prefix=BATCH_PREDICTION_GCS_DEST_PREFIX,\n",
" model_parameters=None,\n",
" machine_type=DEPLOY_COMPUTE,\n",
" accelerator_type=DEPLOY_GPU,\n",
" accelerator_count=DEPLOY_NGPU,\n",
" starting_replica_count=MIN_NODES,\n",
" max_replica_count=MAX_NODES,\n",
" machine_type=\"n1-standard-4\",\n",
" sync=True,\n",
")"
]
@@ -142,176 +142,308 @@
},
"outputs": [],
"source": [
"! pip3 install --upgrade --quiet google-cloud-aiplatform google-cloud-storage pillow numpy\n"
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\") and not os.getenv(\"VIRTUAL_ENV\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG -q\n",
"! pip3 install {USER_FLAG} --upgrade google-cloud-storage -q\n",
"! pip3 install {USER_FLAG} --upgrade pillow -q\n",
"! pip3 install {USER_FLAG} --upgrade numpy -q"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "d98bc9fdd80d"
"id": "restart"
},
"source": [
"### Colab only: Uncomment the following cell to restart the kernel."
"### Restart the kernel\n",
"\n",
"Once you've installed everything, you need to restart the notebook kernel so it can find the packages."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "0dbf29389c65"
"id": "bzPxhxS5lugp"
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs so that your environment can access the new packages\n",
"# import IPython\n",
"import os\n",
"\n",
"# app = IPython.Application.instance()\n",
"# app.kernel.do_shutdown(True)"
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "eec0fc7a0963"
"id": "before_you_begin"
},
"source": [
"## Before you begin\n",
"\n",
"### Select a GPU runtime\n",
"\n",
"**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"2. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"\n",
"3. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n",
"\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "project_id"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, try the following:\n",
"* Run `gcloud config list`.\n",
"* Run `gcloud projects list`.\n",
"* See the support page: [Locate the project ID](https://support.google.com/googleapi/answer/7014113)"
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "e3ce64be5527"
"id": "autoset_project_id"
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}\n",
"\n",
"# Set the project id\n",
"! gcloud config set project {PROJECT_ID}"
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "USd_pUT0lugr"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "250cb8c648d5"
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "7f2e7f78a864"
"id": "2aa333eca058"
},
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable used by Vertex AI. Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "369a2258dd59"
"id": "d8b34ef9a3d0"
},
"outputs": [],
"source": [
"REGION = \"us-central1\" # @param {type: \"string\"}"
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "6a94297012d5"
"id": "timestamp"
},
"source": [
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "c-pX32xalugs"
},
"outputs": [],
"source": [
"from datetime import datetime\n",
"\n",
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "gcp_authenticate"
},
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"Depending on your Jupyter environment, you may have to manually authenticate. Follow the relevant instructions below.\n",
"**If you are using Workbench AI Notebooks**, your environment is already\n",
"authenticated. Skip this step.\n",
"\n",
"**1. Vertex AI Workbench**\n",
"* Do nothing as you are already authenticated."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "28e3c4539627"
},
"source": [
"**2. Local JupyterLab instance, uncomment and run:**\n"
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"\n",
"2. Click **Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "fbc9cd30cc4b"
"id": "vF60K5v1lugs"
},
"outputs": [],
"source": [
"# ! gcloud auth login"
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "79efab26ad02"
},
"source": [
"**3. Colab, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "a336a05c6149"
},
"outputs": [],
"source": [
"# from google.colab import auth\n",
"# auth.authenticate_user()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "e3c28b6b796b"
},
"source": [
"**4. Service account or other**\n",
"* See how to grant Cloud Storage permissions to your service account at https://cloud.google.com/storage/docs/gsutil/commands/iam#ch-examples."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "70a42f1033a3"
"id": "bucket:custom"
},
"source": [
"### Create a Cloud Storage bucket\n",
"\n",
"Create a storage bucket to store intermediate artifacts such as datasets."
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"When you submit a training job using the Cloud SDK, you upload a Python package\n",
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
"the code from this package. In this tutorial, Vertex AI also saves the\n",
"trained model that results from your job in the same bucket. Using this model artifact, you can then\n",
"create Vertex AI model and endpoint resources in order to serve\n",
"online predictions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. It must be unique across all\n",
"Cloud Storage buckets."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "7c12c0866590"
"id": "bucket"
},
"outputs": [],
"source": [
"BUCKET_URI = \"gs://your-bucket-name-unique\" # @param {type:\"string\"}"
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "autoset_bucket"
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n",
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "0cb016da6de3"
"id": "create_bucket"
},
"source": [
"**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket."
@@ -321,11 +453,31 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "eddb0b63434c"
"id": "Oz8J0vmSlugt"
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "validate_bucket"
},
"source": [
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "oadE10x2lugu"
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
]
},
{
@@ -334,6 +486,10 @@
"id": "setup_vars"
},
"source": [
"### Set up variables\n",
"\n",
"Next, set up some variables used throughout the tutorial.\n",
"\n",
"### Import libraries and define constants"
]
},
@@ -346,10 +502,9 @@
"outputs": [],
"source": [
"import os\n",
"import sys\n",
"\n",
"import numpy as np\n",
"from google.cloud import aiplatform\n",
"from PIL import Image"
"from google.cloud import aiplatform"
]
},
{
@@ -374,17 +529,56 @@
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_URI)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "accelerators:training,prediction"
},
"source": [
"#### Set hardware accelerators\n",
"\n",
"You can set hardware accelerators for both training and prediction.\n",
"\n",
"Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Tesla K80 GPUs allocated to each VM, you would specify:\n",
"\n",
" (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 4)\n",
"\n",
"See the [locations where accelerators are available](https://cloud.google.com/vertex-ai/docs/general/locations#accelerators).\n",
"\n",
"Otherwise specify `(None, None)` to use a container image to run on a CPU.\n",
"\n",
"*Note*: TensorFlow releases earlier than 2.3 for GPU support fail to load the custom model in this tutorial. This issue is caused by static graph operations that are generated in the serving function. This is a known issue, which is fixed in TensorFlow 2.3. If you encounter this issue with your own custom models, use a container image for TensorFlow 2.3 or later with GPU support."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "xd5PLXDTlugv"
},
"outputs": [],
"source": [
"TRAIN_GPU, TRAIN_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)\n",
"\n",
"DEPLOY_GPU, DEPLOY_NGPU = (aiplatform.gapic.AcceleratorType.NVIDIA_TESLA_K80, 1)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "container:training,prediction"
},
"source": [
"### Set pre-built containers\n",
"#### Set pre-built containers\n",
"\n",
"Vertex AI provides pre-built containers to run training and prediction.\n",
"\n",
"For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers) and [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)"
"Set the pre-built Docker container image for training and prediction.\n",
"\n",
"\n",
"For the latest list, see [Pre-built containers for training](https://cloud.google.com/ai-platform-unified/docs/training/pre-built-containers).\n",
"\n",
"\n",
"For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/ai-platform-unified/docs/predictions/pre-built-containers)."
]
},
{
@@ -395,11 +589,75 @@
},
"outputs": [],
"source": [
"TRAIN_VERSION = \"tf-cpu.2-9\"\n",
"DEPLOY_VERSION = \"tf2-cpu.2-9\"\n",
"TRAIN_VERSION = \"tf-gpu.2-1\"\n",
"DEPLOY_VERSION = \"tf2-gpu.2-1\"\n",
"\n",
"TRAIN_IMAGE = \"us-docker.pkg.dev/vertex-ai/training/{}:latest\".format(TRAIN_VERSION)\n",
"DEPLOY_IMAGE = \"us-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(DEPLOY_VERSION)"
"TRAIN_IMAGE = \"{}-docker.pkg.dev/vertex-ai/training/{}:latest\".format(\n",
" REGION.split(\"-\")[0], TRAIN_VERSION\n",
")\n",
"DEPLOY_IMAGE = \"{}-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(\n",
" REGION.split(\"-\")[0], DEPLOY_VERSION\n",
")\n",
"\n",
"print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n",
"print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "machine:training,prediction"
},
"source": [
"#### Set machine types\n",
"\n",
"Next, set the machine types to use for training and prediction.\n",
"\n",
"- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure your compute resources for training and prediction.\n",
" - `machine type`\n",
" - `n1-standard`: 3.75GB of memory per vCPU\n",
" - `n1-highmem`: 6.5GB of memory per vCPU\n",
" - `n1-highcpu`: 0.9 GB of memory per vCPU\n",
" - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n",
"\n",
"*Note: The following is not supported for training:*\n",
"\n",
" - `standard`: 2 vCPUs\n",
" - `highcpu`: 2, 4 and 8 vCPUs\n",
"\n",
"*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "YAXwbqKKlugv"
},
"outputs": [],
"source": [
"MACHINE_TYPE = \"n1-standard\"\n",
"\n",
"VCPU = \"4\"\n",
"TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n",
"print(\"Train machine type\", TRAIN_COMPUTE)\n",
"\n",
"MACHINE_TYPE = \"n1-standard\"\n",
"\n",
"VCPU = \"4\"\n",
"DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n",
"print(\"Deploy machine type\", DEPLOY_COMPUTE)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "tutorial_start:custom"
},
"source": [
"# Tutorial\n",
"\n",
"Now you are ready to start creating your own custom-trained model with CIFAR10."
]
},
{
@@ -408,17 +666,21 @@
"id": "train_custom_model"
},
"source": [
"# Tutorial\n",
"\n",
"Now you are ready to start creating your own custom-trained model with CIFAR10.\n",
"## Train a model\n",
"\n",
"There are two ways you can train a custom model using a container image:\n",
"\n",
"- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n",
"\n",
"- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model.\n",
"\n",
"- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "train_custom_job_args"
},
"source": [
"### Define the command args for the training script\n",
"\n",
"Prepare the command-line arguments to pass to your training script.\n",
@@ -439,10 +701,13 @@
},
"outputs": [],
"source": [
"JOB_NAME = \"custom_job_unique\"\n",
"JOB_NAME = \"custom_job_\" + TIMESTAMP\n",
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, JOB_NAME)\n",
"\n",
"TRAIN_STRATEGY = \"single\"\n",
"if not TRAIN_NGPU or TRAIN_NGPU < 2:\n",
" TRAIN_STRATEGY = \"single\"\n",
"else:\n",
" TRAIN_STRATEGY = \"mirror\"\n",
"\n",
"EPOCHS = 20\n",
"STEPS = 100\n",
@@ -620,18 +885,30 @@
" display_name=JOB_NAME,\n",
" script_path=\"task.py\",\n",
" container_uri=TRAIN_IMAGE,\n",
" requirements=[\"tensorflow_datasets\"],\n",
" requirements=[\"tensorflow_datasets==1.3.0\"],\n",
" model_serving_container_image_uri=DEPLOY_IMAGE,\n",
")\n",
"\n",
"MODEL_DISPLAY_NAME = \"cifar10_unique\"\n",
"MODEL_DISPLAY_NAME = \"cifar10-\" + TIMESTAMP\n",
"\n",
"# Start the training\n",
"model = job.run(\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" args=CMDARGS,\n",
" replica_count=1,\n",
")"
"if TRAIN_GPU:\n",
" model = job.run(\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" args=CMDARGS,\n",
" replica_count=1,\n",
" machine_type=TRAIN_COMPUTE,\n",
" accelerator_type=TRAIN_GPU.name,\n",
" accelerator_count=TRAIN_NGPU,\n",
" )\n",
"else:\n",
" model = job.run(\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" args=CMDARGS,\n",
" replica_count=1,\n",
" machine_type=TRAIN_COMPUTE,\n",
" accelerator_count=0,\n",
" )"
]
},
{
@@ -685,20 +962,44 @@
},
"outputs": [],
"source": [
"DEPLOYED_NAME = \"cifar10_deployed_unique\"\n",
"DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n",
"\n",
"TRAFFIC_SPLIT = {\"0\": 100}\n",
"\n",
"MIN_NODES = 1\n",
"MAX_NODES = 1\n",
"\n",
"if DEPLOY_GPU:\n",
" endpoint = model.deploy(\n",
" deployed_model_display_name=DEPLOYED_NAME,\n",
" traffic_split=TRAFFIC_SPLIT,\n",
" machine_type=DEPLOY_COMPUTE,\n",
" accelerator_type=DEPLOY_GPU.name,\n",
" accelerator_count=DEPLOY_NGPU,\n",
" min_replica_count=MIN_NODES,\n",
" max_replica_count=MAX_NODES,\n",
" )\n",
"else:\n",
" endpoint = model.deploy(\n",
" deployed_model_display_name=DEPLOYED_NAME,\n",
" traffic_split=TRAFFIC_SPLIT,\n",
" machine_type=DEPLOY_COMPUTE,\n",
" accelerator_type=DEPLOY_COMPUTE.name,\n",
" accelerator_count=0,\n",
" min_replica_count=MIN_NODES,\n",
" max_replica_count=MAX_NODES,\n",
" )"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "make_prediction"
},
"source": [
"## Make an online prediction request\n",
"\n",
"endpoint = model.deploy(\n",
" deployed_model_display_name=DEPLOYED_NAME,\n",
" traffic_split=TRAFFIC_SPLIT,\n",
" min_replica_count=MIN_NODES,\n",
" max_replica_count=MAX_NODES,\n",
")"
"Send an online prediction request to your deployed model."
]
},
{
@@ -707,9 +1008,6 @@
"id": "get_test_item:test"
},
"source": [
"## Make an online prediction request\n",
"\n",
"Send an online prediction request to your deployed model.\n",
"### Get test data\n",
"\n",
"Download images from the CIFAR dataset and preprocess them.\n",
@@ -755,6 +1053,9 @@
},
"outputs": [],
"source": [
"import numpy as np\n",
"from PIL import Image\n",
"\n",
"# Load image data\n",
"IMAGE_DIRECTORY = \"cifar_test_images\"\n",
"\n",
@@ -865,6 +1166,10 @@
},
"outputs": [],
"source": [
"delete_training_job = True\n",
"delete_model = True\n",
"delete_endpoint = True\n",
"\n",
"# Warning: Setting this to true will delete everything in your bucket\n",
"delete_bucket = False\n",
"\n",
@@ -959,7 +959,6 @@
" script_path=\"custom/trainer/task.py\",\n",
" container_uri=TRAIN_IMAGE,\n",
" requirements=[\"gcsfs==0.7.1\", \"tensorflow-datasets==4.4\"],\n",
" location=REGION,\n",
")\n",
"\n",
"print(job)"
@@ -29,7 +29,7 @@
"id": "JAPoU8Sm5E6e"
},
"source": [
"# Using Vertex AI Feature Store with Pandas Dataframe\n",
"# Using Vertex AI Feature Store with pandas DataFrame\n",
"\n",
"<table align=\"left\">\n",
" <td>\n",
@@ -73,17 +73,13 @@
"source": [
"### Objective\n",
"\n",
"In this notebook, you learn how to use `Vertex AI Feature Store` with pandas Dataframe.\n",
"\n",
"This tutorial uses the following Google Cloud ML services and resources:\n",
"\n",
"- Vertex AI Feature Store\n",
"In this notebook, you learn how to use `Vertex AI Feature Store` with pandas DataFrame.\n",
"\n",
"The steps performed include:\n",
"\n",
"- Ingest Feature values from Pandas DataFrame into Feature Store's Entity types.\n",
"- Read Entity feature values from Online Feature Store into Pandas DataFrame.\n",
"- Batch serve feature values from your Feature Store into Pandas DataFrame.\n",
"- Read Entity Feature values from Online Feature Store into Pandas DataFrame.\n",
"- Batch serve Feature values from your Feature Store into Pandas DataFrame.\n",
"\n",
"You also learn how Vertex AI Feature Store can be useful in the below scenarios:\n",
"\n",
@@ -99,7 +95,7 @@
"source": [
"### Dataset\n",
"\n",
"This tutorial is a part of the Feature Store tutorial notebooks. It uses a movie recommendation dataset as an example for demonstrating various functionalities of Feature Store. The original task is to train a model to predict if a user is going to watch a movie, and serve the model online."
"This tutorial uses a movie recommendation dataset as an example throughout all the notebooks including this one. The original task is to train a model to predict if a user is going to watch a movie and serve the model online."
]
},
{
@@ -150,15 +146,12 @@
"source": [
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"# The Google Cloud Notebook product has specific requirements\n",
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"# Google Cloud Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
"if IS_GOOGLE_CLOUD_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
" \n",
"! pip install -U {USER_FLAG} --upgrade google-cloud-aiplatform \\\n",
@@ -215,7 +208,7 @@
"\n",
"1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n",
"\n",
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
@@ -280,7 +273,7 @@
"#### Region\n",
"\n",
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. It is recommended that you choose the region closest to you.\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
@@ -288,7 +281,7 @@
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
]
},
{
@@ -311,29 +304,10 @@
"id": "dr--iN2kAylZ"
},
"source": [
"#### UUID\n",
"### Authenticate your Google Cloud account\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "4e166d927e36"
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"**If you are using Google Cloud Notebooks**, your environment is already\n",
"authenticated. Skip this step."
]
},
{
@@ -342,11 +316,6 @@
"id": "sBCra4QMA2wR"
},
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
"authenticated. \n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"\n",
@@ -364,7 +333,7 @@
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"5. Click **Create**. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
@@ -379,19 +348,19 @@
},
"outputs": [],
"source": [
"import os\n",
"import sys\n",
"\n",
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"import os\n",
"import sys\n",
"# The Google Cloud Notebook product has specific requirements\n",
"IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
"# If on Google Cloud Notebooks, then don't execute this code\n",
"if not IS_GOOGLE_CLOUD_NOTEBOOK:\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
@@ -401,7 +370,7 @@
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS '[your-service-account-key-path]'"
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
@@ -410,7 +379,7 @@
"id": "XoEqT2Y4DJmf"
},
"source": [
"### Import libraries"
"### Import libraries and define constants"
]
},
{
@@ -424,31 +393,18 @@
"import datetime\n",
"\n",
"import pandas as pd\n",
"from avro.datafile import DataFileReader\n",
"from avro.io import DatumReader\n",
"from google.cloud import aiplatform"
"from google.cloud import aiplatform\n",
"\n",
"aiplatform.init(project=PROJECT_ID, location=REGION)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "138407556b22"
"id": "9UvxYyGUimKw"
},
"source": [
"### Initialize Vertex AI SDK for Python\n",
"\n",
"Initialize the Vertex AI SDK for Python for your project and region."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "8d2077ffee78"
},
"outputs": [],
"source": [
"aiplatform.init(project=PROJECT_ID, location=REGION)"
"## Create Feature Store Resources"
]
},
{
@@ -457,12 +413,11 @@
"id": "buQBIv3ZL3A0"
},
"source": [
"## Create a Feature Store\n",
"### Create Feature Store\n",
"\n",
"The method to create a Feature Store in Vertex AI returns a\n",
"[long-running operation](https://google.aip.dev/151) (LRO). An LRO starts an asynchronous job. LROs are returned for other API methods too, such as updating or deleting a featurestore. \n",
"\n",
"Running the code cell below creates a featurestore and prints the process' logs."
"The method to create a Feature Store returns a\n",
"[long-running operation](https://google.aip.dev/151) (LRO). An LRO starts an asynchronous job. LROs are returned for other API\n",
"methods too, such as updating or deleting a featurestore. Running the code cell creates a featurestore and prints the process logs."
]
},
{
@@ -473,9 +428,9 @@
},
"outputs": [],
"source": [
"# Create featurestore\n",
"movie_predictions_feature_store = aiplatform.Featurestore.create(\n",
" featurestore_id=f\"movie_predictions_{UUID}\", online_store_fixed_node_count=1\n",
" featurestore_id=\"movie_predictions\",\n",
" online_store_fixed_node_count=1,\n",
")"
]
},
@@ -485,7 +440,7 @@
"id": "EpmJq75zXjmT"
},
"source": [
"## Create Entity types\n",
"### Create Entity Types\n",
"\n",
"Entity types can be created within the Featurestore class. Below, you create the `Users` entity type and `Movies` entity type. Process logs are printed in the output for each cell."
]
@@ -498,7 +453,6 @@
},
"outputs": [],
"source": [
"# Create users entity type\n",
"users_entity_type = movie_predictions_feature_store.create_entity_type(\n",
" entity_type_id=\"users\",\n",
" description=\"Users entity\",\n",
@@ -513,7 +467,6 @@
},
"outputs": [],
"source": [
"# Create movies entity type\n",
"movies_entity_type = movie_predictions_feature_store.create_entity_type(\n",
" entity_type_id=\"movies\",\n",
" description=\"Movies entity\",\n",
@@ -526,11 +479,8 @@
"id": "FJW4q-0jO2Xf"
},
"source": [
"## Create Features\n",
"Features can be created within each entity type. Add defined features to the `Users` entity type and `Movies` entity type by using the following methods.\n",
"\n",
"### Add features using *create_feature* method\n",
"Provide the feature information like id, type and description to the `create_feature` method of entity type."
"### Create Features\n",
"Features can be created within each entity type. Add defining features to the `Users` entity type and `Movies` entity type by using the following methods."
]
},
{
@@ -541,21 +491,18 @@
},
"outputs": [],
"source": [
"# Create age feature\n",
"users_feature_age = users_entity_type.create_feature(\n",
" feature_id=\"age\",\n",
" value_type=\"INT64\",\n",
" description=\"User age\",\n",
")\n",
"\n",
"# Create gender feature\n",
"users_feature_gender = users_entity_type.create_feature(\n",
" feature_id=\"gender\",\n",
" value_type=\"STRING\",\n",
" description=\"User gender\",\n",
")\n",
"\n",
"# Create liked_genres feature\n",
"users_feature_liked_genres = users_entity_type.create_feature(\n",
" feature_id=\"liked_genres\",\n",
" value_type=\"STRING_ARRAY\",\n",
@@ -563,18 +510,6 @@
")"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ecb141839033"
},
"source": [
"### Add features using batch method\n",
"You can also create features using a config map in a dictionary format and the `batch_create_features` method. This way, you can add multiple features at once. \n",
"\n",
"Below, you define and create *title*, *genres* and *average_rating* features using the batch method."
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -596,8 +531,17 @@
" \"value_type\": \"DOUBLE\",\n",
" \"description\": \"The average rating for the movie, range is [1.0-5.0]\",\n",
" },\n",
"}\n",
"\n",
"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "YhfOKJL_BvuM"
},
"outputs": [],
"source": [
"movie_features = movies_entity_type.batch_create_features(\n",
" feature_configs=movies_feature_configs,\n",
")"
@@ -609,15 +553,18 @@
"id": "K3n5XdK8Xjmw"
},
"source": [
"## Ingest Feature values into Entity types from dataframes\n",
"## Ingest Feature Values into Entity Type from a Pandas DataFrame\n",
"\n",
"You need to ingest feature values into your entity type containing the features. It is so that you can later `read` (online) or `batch serve` (offline) the feature values from the entity type. \n",
"\n",
"In this step, you learn how to ingest feature values from a Pandas dataframe into an entity type. You can also import feature values from BigQuery or Google Cloud Storage.\n",
"\n",
"### Get data from source\n",
"\n",
"Define the public data sources for users and movies and copy them locally into *avro* files."
"You need to ingest feature values into your entity type containing the features, so you can later `read` (online) or `batch serve` (offline) the feature values from the entity type. In this step, you will learn how to ingest feature values from a Pandas DataFrame into an entity type. We can also import feature values from BigQuery or Google Cloud Storage.\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "BlqJ-QdTcs6W"
},
"source": [
"#### Get data from source files"
]
},
{
@@ -636,8 +583,17 @@
")\n",
"\n",
"USERS_AVRO_FN = \"users.avro\"\n",
"MOVIES_AVRO_FN = \"movies.avro\"\n",
"\n",
"MOVIES_AVRO_FN = \"movies.avro\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "KqIH_bS-5OW5"
},
"outputs": [],
"source": [
"! gsutil cp $GCS_USERS_AVRO_URI $USERS_AVRO_FN\n",
"! gsutil cp $GCS_MOVIES_AVRO_URI $MOVIES_AVRO_FN"
]
@@ -648,9 +604,7 @@
"id": "Fd6Z0jfR5OW5"
},
"source": [
"### Load data from avro files \n",
"\n",
"Load users and movies data from avro files into Pandas dataframes."
"#### Load Avro Files into Pandas DataFrames"
]
},
{
@@ -661,7 +615,10 @@
},
"outputs": [],
"source": [
"# Define a class for reading the avro data\n",
"from avro.datafile import DataFileReader\n",
"from avro.io import DatumReader\n",
"\n",
"\n",
"class AvroReader:\n",
" def __init__(self, data_file):\n",
" self.avro_reader = DataFileReader(open(data_file, \"rb\"), DatumReader())\n",
@@ -679,7 +636,6 @@
},
"outputs": [],
"source": [
"# Load users data from avro file\n",
"users_avro_reader = AvroReader(data_file=USERS_AVRO_FN)\n",
"users_source_df = users_avro_reader.to_dataframe()\n",
"print(users_source_df)"
@@ -693,7 +649,6 @@
},
"outputs": [],
"source": [
"# Load movies data from avro file\n",
"movies_avro_reader = AvroReader(data_file=MOVIES_AVRO_FN)\n",
"movies_source_df = movies_avro_reader.to_dataframe()\n",
"print(movies_source_df)"
@@ -705,9 +660,7 @@
"id": "bgb0WGwX5OW6"
},
"source": [
"### Ingest Feature values into Entity types\n",
"\n",
"Load the feature values into `users` entity type providing the id fields and time field."
"#### Ingest Feature Values into _Users_ Entity Type"
]
},
{
@@ -732,7 +685,7 @@
"id": "PCAdQ3cF5OW6"
},
"source": [
"Load the feature values into `movie` entity type providing the id fields and time field."
"#### Ingest Feature Values into _Movies_ Entity Type"
]
},
{
@@ -757,12 +710,10 @@
"id": "pIYLZwao5OW6"
},
"source": [
"## Read/serve Entity's feature values online from Feature Store\n",
"## Read/Online Serve Entity's Feature Values from Vertex AI Online Feature Store\n",
"\n",
"Feature Store allows [online serving](https://cloud.google.com/vertex-ai/docs/featurestore/serving-online)\n",
"which lets you read feature values for small batches of entities. It works well when you want to read values of selected features from an entity or multiple entities in an entity type.\n",
"\n",
"### Read feature values for users"
"which lets you read feature values for small batches of entities. It works well when you want to read values of selected features from an entity or multiple entities in an entity type."
]
},
{
@@ -779,15 +730,6 @@
"print(users_read_df)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "b2cfa09ef11d"
},
"source": [
"### Read feature values for movies"
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -809,13 +751,18 @@
"id": "AK2Glzkq5OW7"
},
"source": [
"## Batch serve feature values from Feature Store\n",
"## Batch Serve Feature Values from Vertex AI Feature Store\n",
"\n",
"Batch Serving is used to fetch a large batch of feature values for high-throughput, and is typically used for training a model or batch prediction. In this section, you learn how to prepare training examples by using the Feature Store's batch serve function.\n",
"\n",
"### Read instances from source file\n",
"\n",
"Define the source file and destination file. "
"Batch Serving is used to fetch a large batch of feature values for high-throughput, and is typically used for training a model or batch prediction. In this section, you learn how to prepare training examples by using the Feature Store's batch serve function."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "hxsotHUe5OW7"
},
"source": [
"#### Read instances from source file"
]
},
{
@@ -830,15 +777,6 @@
"READ_INSTANCES_CSV_FN = \"data.csv\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "3f2c558b649f"
},
"source": [
"Copy the instances from the source file to the destination file locally."
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -856,9 +794,7 @@
"id": "T5DW1MFt5OW7"
},
"source": [
"### Load the instances\n",
"\n",
"Load the instances from CSV file into a Pandas dataframe."
"#### Load CSV file into a Pandas DataFrame"
]
},
{
@@ -879,9 +815,7 @@
"id": "LsgNNH8G5OW8"
},
"source": [
"### Change the data type\n",
"\n",
"Change the data type of the timestamp field from `Timestamp` to `Datetime64`."
"#### Change the Dtype of `Timestamp` to `Datetime64`"
]
},
{
@@ -903,9 +837,7 @@
"id": "ao1dC5Pc5OW8"
},
"source": [
"### Batch serve feature values from Feature Store\n",
"\n",
"Serve the batch response to a dataframe and display the data."
"#### Batch Serve Feature Values from Movie Predictions Feature Store"
]
},
{
@@ -926,21 +858,43 @@
"movie_predictions_df"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "29gLNORP5OW8"
},
"source": [
"## Read the Updated Feature Values"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "XN84znoI5OW8"
},
"source": [
"## Read the latest feature values\n",
"\n",
"In Feature Store, you access the latest or the last available feature values unless a specific time is provided. Now, you test this feature by ingesting new data to the entity types and reading it from the Feature Store.\n",
"\n",
"### Ingest updated feature values\n",
"\n",
"Now, you update the feature values by running the following cell. \n",
"\n",
"**Note:** For comparison, you can try printing the feature values read from the entity types earlier (those in `movies_read_df` variable). "
"#### Feature Values from last ingestion\n",
"Recall read from the Entity Type shows Feature Values from the last ingestion."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "wtmshq_n5OW9"
},
"outputs": [],
"source": [
"print(movies_read_df)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "feTUJjqG5OW9"
},
"source": [
"#### Ingest updated Feature Values"
]
},
{
@@ -951,13 +905,21 @@
},
"outputs": [],
"source": [
"# Create a dataframe for the new data\n",
"update_movies_df = pd.DataFrame(\n",
" data=[[\"movie_03\", 4.3], [\"movie_04\", 4.8]],\n",
" columns=[\"movie_id\", \"average_rating\"],\n",
")\n",
"\n",
"# Ingest the new data from the dataframe\n",
"print(update_movies_df)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "aKKhSzUc5OW9"
},
"outputs": [],
"source": [
"movies_entity_type.ingest_from_df(\n",
" feature_ids=[\"average_rating\"],\n",
" feature_time=datetime.datetime.now(),\n",
@@ -972,9 +934,8 @@
"id": "s47WCIvL5OW9"
},
"source": [
"### Fetch the latest feature values\n",
"\n",
"Reading from the entity type gives you the updated feature values from the latest ingestion."
"#### Latest Feature Values\n",
"Read from the Entity Type shows updated Feature values from the latest ingestion."
]
},
{
@@ -992,18 +953,23 @@
"print(update_movies_read_df)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "wsvCRzn_5OW9"
},
"source": [
"## Point-in-Time Correctness"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "R1YGRNsW5OW9"
},
"source": [
"## Point-in-time correctness\n",
"\n",
"Vertex AI Feature Store captures feature values for a feature at a specific point in time. In case there are missing values in your past data, you can backfill them using batch serving.\n",
"\n",
"### Missing data\n",
"Recall that response from the batch serve from last ingestion has some missing data in it."
"#### Missing data\n",
"Recall Batch Serve from the last ingestion has some missing data in it."
]
},
{
@@ -1014,7 +980,6 @@
},
"outputs": [],
"source": [
"# Print the response\n",
"print(movie_predictions_df)"
]
},
@@ -1024,9 +989,7 @@
"id": "abQRF6mx5OW-"
},
"source": [
"### Backfill/correct point-in-time data\n",
"\n",
"Impute the missing data based on the timestamps."
"#### Backfill/Correct point-in-time data"
]
},
{
@@ -1037,7 +1000,6 @@
},
"outputs": [],
"source": [
"# Impute the users data\n",
"backfill_users_df = pd.DataFrame(\n",
" data=[[\"bob\", 34, \"Male\", [\"Drama\"], \"2020-02-13 09:35:15\"]],\n",
" columns=[\"user_id\", \"age\", \"gender\", \"liked_genres\", \"update_time\"],\n",
@@ -1054,7 +1016,6 @@
},
"outputs": [],
"source": [
"# Impute the movies data\n",
"backfill_movies_df = pd.DataFrame(\n",
" data=[[\"movie_04\", 4.2, \"The Dark Knight\", \"Action\", \"2020-02-13 09:35:15\"]],\n",
" columns=[\"movie_id\", \"average_rating\", \"title\", \"genres\", \"update_time\"],\n",
@@ -1069,9 +1030,7 @@
"id": "WXb4JUhu5OW-"
},
"source": [
"### Ingest the backfilled/corrected data\n",
"\n",
"Ingest the imputed point-in-time data from dataframe to the entity types in feature store."
"#### Ingest backfilled/corrected point-in-time data from dataframe"
]
},
{
@@ -1082,7 +1041,6 @@
},
"outputs": [],
"source": [
"# Ingest the users data\n",
"users_entity_type.ingest_from_df(\n",
" feature_ids=[\"age\", \"gender\", \"liked_genres\"],\n",
" feature_time=\"update_time\",\n",
@@ -1099,7 +1057,6 @@
},
"outputs": [],
"source": [
"# Ingest the users data\n",
"movies_entity_type.ingest_from_df(\n",
" feature_ids=[\"average_rating\", \"title\", \"genres\"],\n",
" feature_time=\"update_time\",\n",
@@ -1114,8 +1071,8 @@
"id": "1e62Ku6W5OW_"
},
"source": [
"### Fetch the latest data\n",
"Batch serve the latest ingested data with backfill/correction to a dataframe to ensure the feature store is updated. "
"#### Latest ingestion with imputed missing data\n",
"Batch Serve from the latest ingestion with backfill/correction has reduced missing data."
]
},
{
@@ -1147,7 +1104,7 @@
"To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n",
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
"\n",
"Otherwise, you can delete the individual resources you created in this tutorial:"
"You can also keep the project but delete the featurestore by running the code below:"
]
},
{
@@ -1158,7 +1115,6 @@
},
"outputs": [],
"source": [
"# Delete the feature store\n",
"movie_predictions_feature_store.delete(force=True)"
]
}
+27 -19
View File
@@ -1,6 +1,5 @@
## Vertex-AI: Matching Engine Notebook
<a id="sdk_matching_engine_for_indexing"></a>[Create Vertex AI Matching Engine index](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/matching_engine/sdk_matching_engine_for_indexing.ipynb)
[Create Vertex AI Matching Engine index](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/matching_engine/sdk_matching_engine_for_indexing.ipynb)
Learn how to create Approximate Nearest Neighbor (ANN) Index, query against indexes, and validate the performance of the index.
@@ -12,20 +11,29 @@ The steps performed include:
* Perform online query
* Compute recall
<details>
<summary>Example code snippet from the Notebook:</summary>
* Create an IndexEndpoint with VPC Network
```python
# [START aiplatform_sdk_matching_engine_for_indexing]
VPC_NETWORK = "[your-network-name]"
VPC_NETWORK_FULL = "projects/{}/global/networks/{}".format(PROJECT_NUMBER, VPC_NETWORK)
my_index_endpoint = aiplatform.MatchingEngineIndexEndpoint.create(
display_name="index_endpoint_for_demo",
description="index endpoint description",
network=VPC_NETWORK_FULL,
)
# [END aiplatform_sdk_matching_engine_for_indexing]
```
[:notebook: sdk_matching_engine_for_indexing.ipynb](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/matching_engine/sdk_matching_engine_for_indexing.ipynb)
</details>
[Introduction to builtin Swivel embedding algorithm](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/matching_engine/intro-swivel.ipynb)
Learn how to train custom embeddings using Vertex AI Pipelines and deploy the model for serving.
The steps performed include:
1. **Setup**: Importing the required libraries and setting your global variables.
2. **Configure parameters**: Setting the appropriate parameter values for the pipeline job.
3. **Train on Vertex AI Pipelines**: Create a Swivel job to Vertex Pipelines using pipeline template.
4. **Deploy on Vertex AI Prediction**: Importing and deploying the trained model to a callable endpoint.
5. **Predict**: Calling the deployed endpoint using online prediction.
6. **Cleaning up**: Deleting resources created by this tutorial.
[Introduction to builtin Two-towers embedding algorithm](https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/matching_engine/two-tower-model-introduction.ipynb)
Learn how to run the two-tower model.
The steps performed include:
1. **Setup**: Importing the required libraries and setting your global variables.
2. **Configure parameters**: Setting the appropriate parameter values for the training job.
3. **Train on Vertex AI Training**: Submitting a training job.
4. **Deploy on Vertex AI Prediction**: Importing and deploying the trained model to a callable endpoint.
5. **Predict**: Calling the deployed endpoint using online or batch prediction.
6. **Hyperparameter tuning**: Running a hyperparameter tuning job.
7. **Cleaning up**: Deleting resources created by this tutorial.
@@ -758,7 +758,7 @@
"outputs": [],
"source": [
"tree_ah_index = aiplatform.MatchingEngineIndex.create_tree_ah_index(\n",
" display_name=DISPLAY_NAME_BRUTE_FORCE,\n",
" display_name=DISPLAY_NAME,\n",
" contents_delta_uri=EMBEDDINGS_INITIAL_URI,\n",
" dimensions=DIMENSIONS,\n",
" approximate_neighbors_count=150,\n",
@@ -33,67 +33,20 @@
"\n",
"<table align=\"left\">\n",
" <td>\n",
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/migration/UJ10%20Vertex%20SDK%20Custom%20Scikit-Learn%20with%20pre-built%20training%20container.ipynb\">\n",
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/ai-platform-samples/blob/master/vertex-ai-samples/tree/master/notebooks/official/migration/UJ10%20Vertex%20SDK%20Custom%20Scikit-Learn%20with%20pre-built%20training%20container.ipynb\">\n",
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/migration/UJ10%20Vertex%20SDK%20Custom%20Scikit-Learn%20with%20pre-built%20training%20container.ipynb\">\n",
" <a href=\"https://github.com/GoogleCloudPlatform/ai-platform-samples/blob/master/vertex-ai-samples/tree/master/notebooks/official/migration/UJ10%20Vertex%20SDK%20Custom%20Scikit-Learn%20with%20pre-built%20training%20container.ipynb\">\n",
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
" View on GitHub\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/official/migration/UJ10%20Vertex%20SDK%20Custom%20Scikit-Learn%20with%20pre-built%20training%20container.ipynb\">\n",
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
" Open in Vertex AI Workbench\n",
" </a>\n",
" </td> \n",
"</table>\n",
"<br/><br/><br/>"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "7a8a13b86a8b"
},
"source": [
"## Overview\n",
"\n",
"\n",
"This tutorial demonstrates how to use the Vertex AI SDK for Python to train and deploy a custom tabular classification scikit-learn model for batch prediction."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "618cfedf829a"
},
"source": [
"### Objective\n",
"\n",
"In this tutorial, you learn to use `Vertex AI Training` to create a custom trained model and use `Vertex AI Batch Prediction` to do a batch prediction on the trained model.\n",
"\n",
"\n",
"You learn how to create a custom-trained model from a Python script in a Docker container using the Vertex AI SDK for Python, and then do a prediction on the deployed model by sending data.\n",
"\n",
"This tutorial uses the following Google Cloud ML services:\n",
"\n",
"- `Vertex AI Training`\n",
"- `Vertex AI Batch Prediction`\n",
"- `Vertex AI Model` resource\n",
"- `Vertex AI Endpoint` resource\n",
"\n",
"The steps performed include:\n",
"\n",
"- Create a `Vertex AI` custom job for training a scikit-learn model.\n",
"- Upload the trained model artifacts as a `Model` resource.\n",
"- Make a batch prediction.\n",
"- Deploy model to a endpoint\n",
"- Make a online prediction"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -102,7 +55,7 @@
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the UCI Machine Learning [US Census Data (1990) dataset](https://archive.ics.uci.edu/ml/datasets/US+Census+Data+(1990)).The version of the dataset you use in this tutorial is stored in a public Cloud Storage bucket.\n",
"The dataset used for this tutorial is the UCI Machine Learning [US Census Data (1990) dataset](https://archive.ics.uci.edu/ml/datasets/US+Census+Data+(1990)).The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket.\n",
"\n",
"The dataset predicts whether a persons income will be above $50K USD."
]
@@ -135,37 +88,29 @@
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Vertex AI Workbench Notebooks**, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"If you are using Colab or Google Cloud Notebooks, your environment already meets all the requirements to run this notebook. You can skip this step.\n",
"\n",
"**_NOTE_**: This notebook has been tested in the following environment:\n",
"Otherwise, make sure your environment meets this notebook's requirements. You need the following:\n",
"\n",
"* Python version = 3.9\n",
"- The Cloud Storage SDK\n",
"- Git\n",
"- Python 3\n",
"- virtualenv\n",
"- Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
"You need the following:\n",
"The Cloud Storage guide to [Setting up a Python development environment](https://cloud.google.com/python/setup) and the [Jupyter installation guide](https://jupyter.org/install) provide detailed instructions for meeting these requirements. The following steps provide a condensed set of instructions:\n",
"\n",
"* The Google Cloud SDK\n",
"1. [Install and initialize the SDK](https://cloud.google.com/sdk/docs/).\n",
"\n",
"The Google Cloud guide to [Setting up a Python development\n",
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
"for meeting these requirements. The following steps provide a condensed set of\n",
"instructions:\n",
"2. [Install Python 3](https://cloud.google.com/python/setup#installing_python).\n",
"\n",
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
"3. [Install virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv) and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
"4. To install Jupyter, run `pip3 install jupyter` on the command-line in a terminal shell.\n",
"\n",
"1. [Install\n",
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"5. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"1. To install Jupyter, run `pip3 install jupyter` on the\n",
"command-line in a terminal shell.\n",
"\n",
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"1. Open this notebook in the Jupyter Notebook Dashboard."
"6. Open this notebook in the Jupyter Notebook Dashboard.\n"
]
},
{
@@ -176,7 +121,7 @@
"source": [
"## Installation\n",
"\n",
"Install the following packages required to execute this notebook. "
"Install the latest version of Vertex SDK for Python."
]
},
{
@@ -189,18 +134,45 @@
"source": [
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
"# Google Cloud Notebook\n",
"if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" USER_FLAG = \"--user\"\n",
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform google-cloud-storage tensorflow $USER_FLAG -q"
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_storage"
},
"source": [
"Install the latest GA version of *google-cloud-storage* library as well."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_storage"
},
"outputs": [],
"source": [
"! pip3 install -U google-cloud-storage $USER_FLAG"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_tensorflow"
},
"outputs": [],
"source": [
"if os.getenv(\"IS_TESTING\"):\n",
" ! pip3 install --upgrade tensorflow $USER_FLAG"
]
},
{
@@ -222,7 +194,6 @@
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs\n",
"import os\n",
"\n",
"if not os.getenv(\"IS_TESTING\"):\n",
@@ -241,33 +212,26 @@
"source": [
"## Before you begin\n",
"\n",
"### GPU runtime\n",
"\n",
"This tutorial does not require a GPU runtime.\n",
"\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n",
"\n",
"1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com). \n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "5aee4379e8e5"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
@@ -335,10 +299,7 @@
},
"outputs": [],
"source": [
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
"REGION = \"us-central1\" # @param {type: \"string\"}"
]
},
{
@@ -347,9 +308,9 @@
"id": "timestamp"
},
"source": [
"#### UUID\n",
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
]
},
{
@@ -360,16 +321,9 @@
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
@@ -380,31 +334,23 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
"authenticated. \n",
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"2. Click **Create service account**.\n",
"**Click Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
@@ -423,11 +369,8 @@
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
@@ -437,7 +380,7 @@
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS '[your-service-account-key-path]'"
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
@@ -463,8 +406,7 @@
},
"outputs": [],
"source": [
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
]
},
{
@@ -475,9 +417,8 @@
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
]
},
{
@@ -497,7 +438,7 @@
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_NAME"
]
},
{
@@ -517,7 +458,7 @@
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
"! gsutil ls -al $BUCKET_NAME"
]
},
{
@@ -526,6 +467,9 @@
"id": "setup_vars"
},
"source": [
"### Set up variables\n",
"\n",
"Next, set up some variables used throughout the tutorial.\n",
"### Import libraries and define constants"
]
},
@@ -537,11 +481,7 @@
},
"outputs": [],
"source": [
"import json\n",
"import os\n",
"\n",
"import google.cloud.aiplatform as aip\n",
"import tensorflow as tf"
"import google.cloud.aiplatform as aip"
]
},
{
@@ -563,7 +503,7 @@
},
"outputs": [],
"source": [
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
]
},
{
@@ -592,13 +532,10 @@
"outputs": [],
"source": [
"TRAIN_VERSION = \"scikit-learn-cpu.0-23\"\n",
"DEPLOY_VERSION = \"sklearn-cpu.1-0\"\n",
"DEPLOY_VERSION = \"sklearn-cpu.0-23\"\n",
"\n",
"TRAIN_IMAGE = \"us-docker.pkg.dev/vertex-ai/training/{}:latest\".format(TRAIN_VERSION)\n",
"DEPLOY_IMAGE = \"us-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(DEPLOY_VERSION)\n",
"\n",
"print(\"Training:\", TRAIN_IMAGE)\n",
"print(\"Deployment:\", DEPLOY_IMAGE)"
"TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n",
"DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)"
]
},
{
@@ -611,7 +548,7 @@
"\n",
"Next, set the machine type to use for training and prediction.\n",
"\n",
"- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you use for for training and prediction.\n",
"- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n",
" - `machine type`\n",
" - `n1-standard`: 3.75GB of memory per vCPU.\n",
" - `n1-highmem`: 6.5GB of memory per vCPU\n",
@@ -663,7 +600,7 @@
"\n",
"#### Package layout\n",
"\n",
"Before you start the training, you look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n",
"Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n",
"\n",
"- PKG-INFO\n",
"- README.md\n",
@@ -679,7 +616,7 @@
"\n",
"#### Package Assembly\n",
"\n",
"In the following cells, you assemble the training package."
"In the following cells, you will assemble the training package."
]
},
{
@@ -700,7 +637,7 @@
"setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n",
"! echo \"$setup_cfg\" > custom/setup.cfg\n",
"\n",
"setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n",
"setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n",
"! echo \"$setup_py\" > custom/setup.py\n",
"\n",
"pkg_info = \"Metadata-Version: 1.0\\n\\nName: US Census Data (1990) tabular binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n",
@@ -742,6 +679,7 @@
"parser.add_argument('--model-dir', dest='model_dir',\n",
" default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n",
"args = parser.parse_args()\n",
"\n",
"print('Python Version = {}'.format(sys.version))\n",
"\n",
"# Public bucket holding the census data\n",
@@ -857,9 +795,6 @@
"subdirs = args.model_dir.split('/')[3:]\n",
"subdir = subdirs[0]\n",
"subdirs.pop(0)\n",
"\n",
"\n",
"\n",
"for comp in subdirs:\n",
" subdir = os.path.join(subdir, comp)\n",
"\n",
@@ -868,7 +803,7 @@
"\n",
"# Upload the model to GCS\n",
"bucket = storage.Client().bucket(bucket)\n",
"blob = bucket.blob(subdir + 'model.joblib')\n",
"blob = bucket.blob(subdir + '/model.joblib')\n",
"blob.upload_from_filename('model.joblib')"
]
},
@@ -894,7 +829,7 @@
"! rm -f custom.tar custom.tar.gz\n",
"! tar cvf custom.tar custom\n",
"! gzip custom.tar\n",
"! gsutil cp custom.tar.gz $BUCKET_URI/trainer_census.tar.gz"
"! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_census.tar.gz"
]
},
{
@@ -945,10 +880,10 @@
"outputs": [],
"source": [
"job = aip.CustomTrainingJob(\n",
" display_name=\"census_\" + UUID,\n",
" display_name=\"census_\" + TIMESTAMP,\n",
" script_path=\"custom/trainer/task.py\",\n",
" container_uri=TRAIN_IMAGE,\n",
" requirements=[\"gcsfs\", \"tensorflow-datasets\"],\n",
" requirements=[\"gcsfs==0.7.1\", \"tensorflow-datasets==4.4\"],\n",
")\n",
"\n",
"print(job)"
@@ -989,7 +924,7 @@
},
"outputs": [],
"source": [
"MODEL_DIR = \"{}/{}\".format(BUCKET_URI, UUID)\n",
"MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, TIMESTAMP)\n",
"\n",
"\n",
"job.run(\n",
@@ -1036,7 +971,7 @@
"outputs": [],
"source": [
"model = aip.Model.upload(\n",
" display_name=\"census_\" + UUID,\n",
" display_name=\"census_\" + TIMESTAMP,\n",
" artifact_uri=MODEL_DIR,\n",
" serving_container_image_uri=DEPLOY_IMAGE,\n",
" sync=False,\n",
@@ -1086,7 +1021,7 @@
"source": [
"### Make test items\n",
"\n",
"You use synthetic data as test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction."
"You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction."
]
},
{
@@ -1141,7 +1076,7 @@
"source": [
"### Make the batch input file\n",
"\n",
"Now make a batch input file, which you store in your local Cloud Storage bucket. Each instance in the prediction request is a list of the form:\n",
"Now make a batch input file, which you will store in your local Cloud Storage bucket. Each instance in the prediction request is a list of the form:\n",
"\n",
" [ [ content_1], [content_2] ]\n",
"\n",
@@ -1156,7 +1091,11 @@
},
"outputs": [],
"source": [
"gcs_input_uri = BUCKET_URI + \"/\" + \"test.jsonl\"\n",
"import json\n",
"\n",
"import tensorflow as tf\n",
"\n",
"gcs_input_uri = BUCKET_NAME + \"/\" + \"test.jsonl\"\n",
"with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n",
" for i in INSTANCES:\n",
" f.write(json.dumps(i) + \"\\n\")\n",
@@ -1195,9 +1134,9 @@
"MAX_NODES = 1\n",
"\n",
"batch_predict_job = model.batch_predict(\n",
" job_display_name=\"census_\" + UUID,\n",
" job_display_name=\"census_\" + TIMESTAMP,\n",
" gcs_source=gcs_input_uri,\n",
" gcs_destination_prefix=BUCKET_URI,\n",
" gcs_destination_prefix=BUCKET_NAME,\n",
" instances_format=\"jsonl\",\n",
" predictions_format=\"jsonl\",\n",
" model_parameters=None,\n",
@@ -1309,6 +1248,8 @@
},
"outputs": [],
"source": [
"import json\n",
"\n",
"bp_iter_outputs = batch_predict_job.iter_outputs()\n",
"\n",
"prediction_results = list()\n",
@@ -1322,7 +1263,8 @@
" with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n",
" for line in gfile.readlines():\n",
" line = json.loads(line)\n",
" print(line)"
" print(line)\n",
" break"
]
},
{
@@ -1381,7 +1323,7 @@
},
"outputs": [],
"source": [
"DEPLOYED_NAME = \"census-\" + UUID\n",
"DEPLOYED_NAME = \"census-\" + TIMESTAMP\n",
"\n",
"TRAFFIC_SPLIT = {\"0\": 100}\n",
"\n",
@@ -1415,6 +1357,15 @@
" INFO:google.cloud.aiplatform.models:Endpoint model deployed. Resource name: projects/759209241365/locations/us-central1/endpoints/4867177336350441472"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "endpoints_predict:migration,new,mbsdk"
},
"source": [
"### [predictions.online-prediction-automl](https://cloud.google.com/vertex-ai/docs/predictions/online-predictions-automl)"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -1423,7 +1374,7 @@
"source": [
"### Make test item\n",
"\n",
"You use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction."
"You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction."
]
},
{
@@ -1540,10 +1491,13 @@
"Otherwise, you can delete the individual resources you created in this tutorial:\n",
"\n",
"- Dataset\n",
"- Pipeline\n",
"- Model\n",
"- Endpoint\n",
"- AutoML Training Job\n",
"- Batch Job\n",
"- Custom Job\n",
"- Hyperparameter Tuning Job\n",
"- Cloud Storage Bucket"
]
},
@@ -1551,25 +1505,64 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "74ddbf65df4c"
"id": "cleanup:mbsdk"
},
"outputs": [],
"source": [
"# delete endpoint\n",
"endpoint.delete()\n",
"delete_all = True\n",
"\n",
"# Delete the model using the Vertex model object\n",
"model.delete()\n",
"if delete_all:\n",
" # Delete the dataset using the Vertex dataset object\n",
" try:\n",
" if \"dataset\" in globals():\n",
" dataset.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"# Delete the AutoML or Pipeline training job\n",
"job.delete()\n",
" # Delete the model using the Vertex model object\n",
" try:\n",
" if \"model\" in globals():\n",
" model.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"# Delete the batch prediction job using the Vertex batch prediction object\n",
"batch_predict_job.delete()\n",
" # Delete the endpoint using the Vertex endpoint object\n",
" try:\n",
" if \"endpoint\" in globals():\n",
" endpoint.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"delete_bucket = False\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
" ! gsutil -m rm -r $BUCKET_URI"
" # Delete the AutoML or Pipeline trainig job\n",
" try:\n",
" if \"dag\" in globals():\n",
" dag.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" # Delete the custom trainig job\n",
" try:\n",
" if \"job\" in globals():\n",
" job.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" # Delete the batch prediction job using the Vertex batch prediction object\n",
" try:\n",
" if \"batch_predict_job\" in globals():\n",
" batch_predict_job.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
" try:\n",
" if \"hpt_job\" in globals():\n",
" hpt_job.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" if \"BUCKET_NAME\" in globals():\n",
" ! gsutil rm -r $BUCKET_NAME"
]
}
],
@@ -43,58 +43,10 @@
" View on GitHub\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/migration/UJ7%20Vertex%20SDK%20AutoML%20Text%20Entity%20Extraction.ipynb\">\n",
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
" Open in Vertex AI Workbench\n",
" </a>\n",
" </td> \n",
"</table>\n",
"<br/><br/><br/>"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "2277f661a148"
},
"source": [
"## Overview\n",
"\n",
"<a name=\"section-1\"></a>\n",
"\n",
"This notebook demonstrates how to create an AutoML Text Entity Extrasction Model, with a Vertex AI ncbi disease research dataset, and how to serve the model for batch prediction. It requires you provide a bucket where the dataset will be stored.\n",
"\n",
"Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "f926ec7acab3"
},
"source": [
"### Objective\n",
"\n",
"The objective of this notebook is to build a AutoML Text Entity Extrasction Model. The following steps have been followed:\n",
"This tutorial uses the following Google Cloud ML services :\n",
"\n",
"* Vertex AI Dataset resource\n",
"* AutoML Training\n",
"* Vertex AI Model resource\n",
"* Vertex AI Batch Prediction\n",
"\n",
"The steps performed include the following:\n",
"\n",
"* Set your task name, and GCS prefix\n",
"* Copy AutoML video demo train data for creating managed dataset\n",
"* Create a dataset on Vertex AI.\n",
"* Configure a training job\n",
"* Launch a training job and create a model on Vertex AI\n",
"* Copy AutoML Video Demo Prediction Data for creating batch prediction job\n",
"* Perform batch prediction job on the model"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -167,7 +119,7 @@
"source": [
"## Installation\n",
"\n",
"Install the packages required for executing this notebook."
"Install the latest version of Vertex SDK for Python."
]
},
{
@@ -186,9 +138,7 @@
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform \\\n",
" cuda-python \\\n",
" $USER_FLAG -q"
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
@@ -328,7 +278,7 @@
"#### Region\n",
"\n",
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. It is recommended that you choose the region closest to you.\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
@@ -336,7 +286,7 @@
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)"
]
},
{
@@ -347,10 +297,7 @@
},
"outputs": [],
"source": [
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
"REGION = \"us-central1\" # @param {type: \"string\"}"
]
},
{
@@ -359,9 +306,9 @@
"id": "timestamp"
},
"source": [
"#### UUID\n",
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
]
},
{
@@ -372,16 +319,9 @@
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
@@ -392,38 +332,23 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
"authenticated. Skip this step."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "f1d7a972141f"
},
"source": [
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n",
"\n",
"2. Click **Create service account**.\n",
"**Click Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"In the **Service account name** field, enter a name, and click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"Click Create. A JSON file that contains your key downloads to your local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
"Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell."
]
},
{
@@ -442,11 +367,8 @@
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
@@ -469,16 +391,9 @@
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n",
"\n",
"When you submit a training job using the Cloud SDK, you upload a Python package\n",
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
"the code from this package. In this tutorial, Vertex AI also saves the\n",
"trained model that results from your job in the same bucket. Using this model artifact, you can then\n",
"create Vertex AI model and endpoint resources in order to serve\n",
"online predictions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. It must be unique across all\n",
"Cloud Storage buckets."
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization."
]
},
{
@@ -489,8 +404,7 @@
},
"outputs": [],
"source": [
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
]
},
{
@@ -501,9 +415,8 @@
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
]
},
{
@@ -523,7 +436,7 @@
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_NAME"
]
},
{
@@ -543,7 +456,7 @@
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
"! gsutil ls -al $BUCKET_NAME"
]
},
{
@@ -575,7 +488,7 @@
"id": "init_aip:mbsdk"
},
"source": [
"## Initialize Vertex AI SDK for Python\n",
"## Initialize Vertex SDK for Python\n",
"\n",
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
]
@@ -690,7 +603,7 @@
"outputs": [],
"source": [
"dataset = aip.TextDataset.create(\n",
" display_name=\"NCBI Biomedical\" + \"_\" + UUID,\n",
" display_name=\"NCBI Biomedical\" + \"_\" + TIMESTAMP,\n",
" gcs_source=[IMPORT_FILE],\n",
" import_schema_uri=aip.schema.dataset.ioformat.text.extraction,\n",
")\n",
@@ -769,7 +682,7 @@
"outputs": [],
"source": [
"dag = aip.AutoMLTextTrainingJob(\n",
" display_name=\"biomedical_\" + UUID, prediction_type=\"extraction\"\n",
" display_name=\"biomedical_\" + TIMESTAMP, prediction_type=\"extraction\"\n",
")\n",
"\n",
"print(dag)"
@@ -817,7 +730,7 @@
"source": [
"model = dag.run(\n",
" dataset=dataset,\n",
" model_display_name=\"biomedical_\" + UUID,\n",
" model_display_name=\"biomedical_\" + TIMESTAMP,\n",
" training_fraction_split=0.8,\n",
" validation_fraction_split=0.1,\n",
" test_fraction_split=0.1,\n",
@@ -888,7 +801,7 @@
"outputs": [],
"source": [
"# Get model resource ID\n",
"models = aip.Model.list(filter=\"display_name=biomedical_\" + UUID)\n",
"models = aip.Model.list(filter=\"display_name=biomedical_\" + TIMESTAMP)\n",
"\n",
"# Get a reference to the Model Service client\n",
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
@@ -1012,14 +925,14 @@
"\n",
"import tensorflow as tf\n",
"\n",
"gcs_test_item_1 = BUCKET_URI + \"/test1.txt\"\n",
"gcs_test_item_1 = BUCKET_NAME + \"/test1.txt\"\n",
"with tf.io.gfile.GFile(gcs_test_item_1, \"w\") as f:\n",
" f.write(test_item_1 + \"\\n\")\n",
"gcs_test_item_2 = BUCKET_URI + \"/test2.txt\"\n",
"gcs_test_item_2 = BUCKET_NAME + \"/test2.txt\"\n",
"with tf.io.gfile.GFile(gcs_test_item_2, \"w\") as f:\n",
" f.write(test_item_2 + \"\\n\")\n",
"\n",
"gcs_input_uri = BUCKET_URI + \"/test.jsonl\"\n",
"gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n",
"with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n",
" data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n",
" f.write(json.dumps(data) + \"\\n\")\n",
@@ -1055,9 +968,9 @@
"outputs": [],
"source": [
"batch_predict_job = model.batch_predict(\n",
" job_display_name=\"biomedical_\" + UUID,\n",
" job_display_name=\"biomedical_\" + TIMESTAMP,\n",
" gcs_source=gcs_input_uri,\n",
" gcs_destination_prefix=BUCKET_URI,\n",
" gcs_destination_prefix=BUCKET_NAME,\n",
" sync=False,\n",
")\n",
"\n",
@@ -1398,31 +1311,60 @@
},
"outputs": [],
"source": [
"# Delete the dataset using the Vertex dataset object\n",
"delete_all = True\n",
"\n",
"dataset.delete()\n",
"if delete_all:\n",
" # Delete the dataset using the Vertex dataset object\n",
" try:\n",
" if \"dataset\" in globals():\n",
" dataset.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"# Delete the model using the Vertex model object\n",
" # Delete the model using the Vertex model object\n",
" try:\n",
" if \"model\" in globals():\n",
" model.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"model.delete()\n",
" # Delete the endpoint using the Vertex endpoint object\n",
" try:\n",
" if \"endpoint\" in globals():\n",
" endpoint.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"# Delete the endpoint using the Vertex endpoint object\n",
" # Delete the AutoML or Pipeline trainig job\n",
" try:\n",
" if \"dag\" in globals():\n",
" dag.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"endpoint.delete()\n",
"# Delete the AutoML or Pipeline trainig job\n",
" # Delete the custom trainig job\n",
" try:\n",
" if \"job\" in globals():\n",
" job.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"dag.delete()\n",
" # Delete the batch prediction job using the Vertex batch prediction object\n",
" try:\n",
" if \"batch_predict_job\" in globals():\n",
" batch_predict_job.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"# Delete the batch prediction job using the Vertex batch prediction object\n",
" # Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
" try:\n",
" if \"hpt_job\" in globals():\n",
" hpt_job.delete()\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"batch_predict_job.delete()\n",
"\n",
"# Delete the hyperparameter tuning job using the Vertex hyperparameter tuning object\n",
"\n",
"# Delete GCS bucket.\n",
"delete_bucket = False\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
" ! gsutil -m rm -r $BUCKET_URI"
" if \"BUCKET_NAME\" in globals():\n",
" ! gsutil rm -r $BUCKET_NAME"
]
}
],
@@ -3,7 +3,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "bdccc50b",
"metadata": {
"id": "copyright"
},
@@ -26,7 +25,6 @@
},
{
"cell_type": "markdown",
"id": "c6c22009",
"metadata": {
"id": "title:migration,new"
},
@@ -58,19 +56,17 @@
},
{
"cell_type": "markdown",
"id": "b3558cd7",
"metadata": {
"id": "dataset:claritin,tst"
},
"source": [
"### Dataset\n",
"\n",
"The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) from [data.world Datasets](https://data.world). The version of the dataset you use in this tutorial is stored in a public Cloud Storage bucket."
"The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) from [data.world Datasets](https://data.world). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket."
]
},
{
"cell_type": "markdown",
"id": "9b9362da",
"metadata": {
"id": "costs"
},
@@ -91,7 +87,6 @@
},
{
"cell_type": "markdown",
"id": "05425dbe",
"metadata": {
"id": "setup_local"
},
@@ -125,7 +120,6 @@
},
{
"cell_type": "markdown",
"id": "070c64e0",
"metadata": {
"id": "install_aip:mbsdk"
},
@@ -138,7 +132,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "f6b14b99",
"metadata": {
"id": "install_aip:mbsdk"
},
@@ -152,12 +145,42 @@
"else:\n",
" USER_FLAG = \"\"\n",
"\n",
"! pip3 install --upgrade google-cloud-aiplatform google-cloud-storage tensorflow $USER_FLAG -q"
"! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "install_storage"
},
"source": [
"Install the latest GA version of *google-cloud-storage* library as well."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_storage"
},
"outputs": [],
"source": [
"! pip3 install -U google-cloud-storage $USER_FLAG"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "install_tensorflow"
},
"outputs": [],
"source": [
"! pip3 install --upgrade tensorflow $USER_FLAG"
]
},
{
"cell_type": "markdown",
"id": "81f60b84",
"metadata": {
"id": "restart"
},
@@ -170,7 +193,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "a95627f0",
"metadata": {
"id": "restart"
},
@@ -188,7 +210,6 @@
},
{
"cell_type": "markdown",
"id": "b7f6b038",
"metadata": {
"id": "before_you_begin:nogpu"
},
@@ -209,7 +230,7 @@
"\n",
"3. [Enable the following APIs: Vertex AI APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"4. If you are running this notebook locally, you need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK]((https://cloud.google.com/sdk)).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
@@ -220,7 +241,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "f55cca7c",
"metadata": {
"id": "set_project_id"
},
@@ -232,7 +252,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "6917314c",
"metadata": {
"id": "autoset_project_id"
},
@@ -248,7 +267,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "53236aa8",
"metadata": {
"id": "set_gcloud_project_id"
},
@@ -259,7 +277,6 @@
},
{
"cell_type": "markdown",
"id": "c009cc18",
"metadata": {
"id": "region"
},
@@ -281,61 +298,47 @@
{
"cell_type": "code",
"execution_count": null,
"id": "071a11c0",
"metadata": {
"id": "region"
},
"outputs": [],
"source": [
"REGION = \"us-central1\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
"REGION = \"us-central1\" # @param {type: \"string\"}"
]
},
{
"cell_type": "markdown",
"id": "ae48374d",
"metadata": {
"id": "timestamp"
},
"source": [
"#### UUID\n",
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append the timestamp onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "41ba0990",
"metadata": {
"id": "timestamp"
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
"cell_type": "markdown",
"id": "2128e871",
"metadata": {
"id": "gcp_authenticate"
},
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already authenticated.\n",
"**If you are using Google Cloud Notebooks**, your environment is already authenticated. Skip this step.\n",
"\n",
"**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n",
"\n",
@@ -357,7 +360,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "433e860c",
"metadata": {
"id": "gcp_authenticate"
},
@@ -371,11 +373,8 @@
"import os\n",
"import sys\n",
"\n",
"# If on Vertex AI Workbench, then don't execute this code\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\") and not os.getenv(\n",
" \"DL_ANACONDA_HOME\"\n",
"):\n",
"# If on Google Cloud Notebook, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
@@ -390,7 +389,6 @@
},
{
"cell_type": "markdown",
"id": "b57cb5f6",
"metadata": {
"id": "bucket:mbsdk"
},
@@ -407,33 +405,28 @@
{
"cell_type": "code",
"execution_count": null,
"id": "61b082b1",
"metadata": {
"id": "bucket"
},
"outputs": [],
"source": [
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
"BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ff81b3cc",
"metadata": {
"id": "autoset_bucket"
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = \"gs://\" + BUCKET_NAME"
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n",
" BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
]
},
{
"cell_type": "markdown",
"id": "f8c009cd",
"metadata": {
"id": "create_bucket"
},
@@ -444,18 +437,16 @@
{
"cell_type": "code",
"execution_count": null,
"id": "2f881cb5",
"metadata": {
"id": "create_bucket"
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_NAME"
]
},
{
"cell_type": "markdown",
"id": "d746d0f0",
"metadata": {
"id": "validate_bucket"
},
@@ -466,18 +457,16 @@
{
"cell_type": "code",
"execution_count": null,
"id": "8c435668",
"metadata": {
"id": "validate_bucket"
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
"! gsutil ls -al $BUCKET_NAME"
]
},
{
"cell_type": "markdown",
"id": "f578b01b",
"metadata": {
"id": "setup_vars"
},
@@ -491,7 +480,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "f41ecf1e",
"metadata": {
"id": "import_aip:mbsdk"
},
@@ -502,31 +490,28 @@
},
{
"cell_type": "markdown",
"id": "292245fd",
"metadata": {
"id": "init_aip:mbsdk"
},
"source": [
"## Initialize Vertex AI SDK for Python\n",
"## Initialize Vertex SDK for Python\n",
"\n",
"Initialize the Vertex AI SDK for Python for your project and corresponding bucket."
"Initialize the Vertex SDK for Python for your project and corresponding bucket."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "56dc88d8",
"metadata": {
"id": "init_aip:mbsdk"
},
"outputs": [],
"source": [
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_URI)"
"aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)"
]
},
{
"cell_type": "markdown",
"id": "87e20f86",
"metadata": {
"id": "import_file:u_dataset,csv"
},
@@ -539,7 +524,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "fffdec5d",
"metadata": {
"id": "import_file:claritin,csv,tst"
},
@@ -551,7 +535,6 @@
},
{
"cell_type": "markdown",
"id": "9c8d950b",
"metadata": {
"id": "quick_peek:csv"
},
@@ -566,7 +549,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "da6d0980",
"metadata": {
"id": "quick_peek:csv"
},
@@ -586,7 +568,6 @@
},
{
"cell_type": "markdown",
"id": "00e9f81e",
"metadata": {
"id": "create_a_dataset:migration"
},
@@ -596,7 +577,6 @@
},
{
"cell_type": "markdown",
"id": "14b72768",
"metadata": {
"id": "datasets_create:migration,new,mbsdk"
},
@@ -606,7 +586,6 @@
},
{
"cell_type": "markdown",
"id": "00d777bd",
"metadata": {
"id": "create_dataset:text,tst"
},
@@ -625,14 +604,13 @@
{
"cell_type": "code",
"execution_count": null,
"id": "ca8a6f66",
"metadata": {
"id": "create_dataset:text,tst"
},
"outputs": [],
"source": [
"dataset = aip.TextDataset.create(\n",
" display_name=\"Crowdflower Claritin-Twitter\" + \"_\" + UUID,\n",
" display_name=\"Crowdflower Claritin-Twitter\" + \"_\" + TIMESTAMP,\n",
" gcs_source=[IMPORT_FILE],\n",
" import_schema_uri=aip.schema.dataset.ioformat.text.sentiment,\n",
")\n",
@@ -642,7 +620,6 @@
},
{
"cell_type": "markdown",
"id": "068df169",
"metadata": {
"id": "create_dataset:text,tst"
},
@@ -662,7 +639,6 @@
},
{
"cell_type": "markdown",
"id": "fb50a4ce",
"metadata": {
"id": "train_a_model:migration"
},
@@ -672,7 +648,6 @@
},
{
"cell_type": "markdown",
"id": "293160ba",
"metadata": {
"id": "trainingpipelines_create:migration,new,mbsdk"
},
@@ -682,7 +657,6 @@
},
{
"cell_type": "markdown",
"id": "84801634",
"metadata": {
"id": "create_automl_pipeline:text,tst"
},
@@ -709,14 +683,13 @@
{
"cell_type": "code",
"execution_count": null,
"id": "69eaae0e",
"metadata": {
"id": "create_automl_pipeline:text,tst"
},
"outputs": [],
"source": [
"dag = aip.AutoMLTextTrainingJob(\n",
" display_name=\"claritin_\" + UUID,\n",
" display_name=\"claritin_\" + TIMESTAMP,\n",
" prediction_type=\"sentiment\",\n",
" sentiment_max=SENTIMENT_MAX,\n",
")\n",
@@ -726,7 +699,6 @@
},
{
"cell_type": "markdown",
"id": "da9ecb4e",
"metadata": {
"id": "create_automl_pipeline:text,tst"
},
@@ -738,7 +710,6 @@
},
{
"cell_type": "markdown",
"id": "55f19997",
"metadata": {
"id": "run_automl_pipeline:text"
},
@@ -755,13 +726,12 @@
"\n",
"The `run` method when completed returns the `Model` resource.\n",
"\n",
"The execution of the training pipeline take upto 20 minutes."
"The execution of the training pipeline will take upto 20 minutes."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6149074c",
"metadata": {
"id": "run_automl_pipeline:text"
},
@@ -769,7 +739,7 @@
"source": [
"model = dag.run(\n",
" dataset=dataset,\n",
" model_display_name=\"claritin_\" + UUID,\n",
" model_display_name=\"claritin_\" + TIMESTAMP,\n",
" training_fraction_split=0.8,\n",
" validation_fraction_split=0.1,\n",
" test_fraction_split=0.1,\n",
@@ -778,7 +748,6 @@
},
{
"cell_type": "markdown",
"id": "6e8fe148",
"metadata": {
"id": "run_automl_pipeline:text"
},
@@ -804,7 +773,6 @@
},
{
"cell_type": "markdown",
"id": "c25dee28",
"metadata": {
"id": "evaluate_the_model:migration"
},
@@ -814,7 +782,6 @@
},
{
"cell_type": "markdown",
"id": "903e8226",
"metadata": {
"id": "models_evaluations_list:migration,new"
},
@@ -824,7 +791,6 @@
},
{
"cell_type": "markdown",
"id": "cb2d95f3",
"metadata": {
"id": "evaluate_the_model:mbsdk"
},
@@ -838,14 +804,13 @@
{
"cell_type": "code",
"execution_count": null,
"id": "9b1ec312",
"metadata": {
"id": "evaluate_the_model:mbsdk"
},
"outputs": [],
"source": [
"# Get model resource ID\n",
"models = aip.Model.list(filter=\"display_name=claritin_\" + UUID)\n",
"models = aip.Model.list(filter=\"display_name=claritin_\" + TIMESTAMP)\n",
"\n",
"# Get a reference to the Model Service client\n",
"client_options = {\"api_endpoint\": f\"{REGION}-aiplatform.googleapis.com\"}\n",
@@ -860,7 +825,6 @@
},
{
"cell_type": "markdown",
"id": "9eab460e",
"metadata": {
"id": "evaluate_the_model:mbsdk"
},
@@ -901,7 +865,6 @@
},
{
"cell_type": "markdown",
"id": "d4111c50",
"metadata": {
"id": "make_batch_predictions:migration"
},
@@ -911,7 +874,6 @@
},
{
"cell_type": "markdown",
"id": "f73fad68",
"metadata": {
"id": "batchpredictionjobs_create:migration,new,mbsdk"
},
@@ -921,20 +883,18 @@
},
{
"cell_type": "markdown",
"id": "ba77f1c7",
"metadata": {
"id": "get_test_items:batch_prediction"
},
"source": [
"### Get test item(s)\n",
"\n",
"Now do a batch prediction to your Vertex model. You use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction."
"Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "1c9fc91f",
"metadata": {
"id": "get_test_items:automl,tst,csv"
},
@@ -956,14 +916,13 @@
},
{
"cell_type": "markdown",
"id": "2a18c8e2",
"metadata": {
"id": "make_batch_file:automl,text"
},
"source": [
"### Make the batch input file\n",
"\n",
"Now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n",
"Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n",
"\n",
"- `content`: The Cloud Storage path to the file with the text item.\n",
"- `mime_type`: The content type. In our example, it is a `text` file.\n",
@@ -976,7 +935,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "70461c41",
"metadata": {
"id": "make_batch_file:automl,text"
},
@@ -986,14 +944,14 @@
"\n",
"import tensorflow as tf\n",
"\n",
"gcs_test_item_1 = BUCKET_URI + \"/test1.txt\"\n",
"gcs_test_item_1 = BUCKET_NAME + \"/test1.txt\"\n",
"with tf.io.gfile.GFile(gcs_test_item_1, \"w\") as f:\n",
" f.write(test_item_1 + \"\\n\")\n",
"gcs_test_item_2 = BUCKET_URI + \"/test2.txt\"\n",
"gcs_test_item_2 = BUCKET_NAME + \"/test2.txt\"\n",
"with tf.io.gfile.GFile(gcs_test_item_2, \"w\") as f:\n",
" f.write(test_item_2 + \"\\n\")\n",
"\n",
"gcs_input_uri = BUCKET_URI + \"/test.jsonl\"\n",
"gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n",
"with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n",
" data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n",
" f.write(json.dumps(data) + \"\\n\")\n",
@@ -1006,7 +964,6 @@
},
{
"cell_type": "markdown",
"id": "254cbdbb",
"metadata": {
"id": "batch_request:mbsdk"
},
@@ -1018,22 +975,21 @@
"- `job_display_name`: The human readable name for the batch prediction job.\n",
"- `gcs_source`: A list of one or more batch request input files.\n",
"- `gcs_destination_prefix`: The Cloud Storage location for storing the batch prediction resuls.\n",
"- `sync`: If set to True, the call block while waiting for the asynchronous batch job to complete."
"- `sync`: If set to True, the call will block while waiting for the asynchronous batch job to complete."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8f3cf8b6",
"metadata": {
"id": "batch_request:mbsdk"
},
"outputs": [],
"source": [
"batch_predict_job = model.batch_predict(\n",
" job_display_name=\"claritin_\" + UUID,\n",
" job_display_name=\"claritin_\" + TIMESTAMP,\n",
" gcs_source=gcs_input_uri,\n",
" gcs_destination_prefix=BUCKET_URI,\n",
" gcs_destination_prefix=BUCKET_NAME,\n",
" sync=False,\n",
")\n",
"\n",
@@ -1042,7 +998,6 @@
},
{
"cell_type": "markdown",
"id": "530dbf5b",
"metadata": {
"id": "batch_request:mbsdk"
},
@@ -1062,7 +1017,6 @@
},
{
"cell_type": "markdown",
"id": "89414481",
"metadata": {
"id": "batch_request_wait:mbsdk"
},
@@ -1075,7 +1029,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "a579bd4a",
"metadata": {
"id": "batch_request_wait:mbsdk"
},
@@ -1086,7 +1039,6 @@
},
{
"cell_type": "markdown",
"id": "2cba4cc6",
"metadata": {
"id": "batch_request_wait:mbsdk"
},
@@ -1121,7 +1073,6 @@
},
{
"cell_type": "markdown",
"id": "c46e3e76",
"metadata": {
"id": "get_batch_prediction:mbsdk,tst"
},
@@ -1140,7 +1091,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "d2af5ea8",
"metadata": {
"id": "get_batch_prediction:mbsdk,tst"
},
@@ -1169,7 +1119,6 @@
},
{
"cell_type": "markdown",
"id": "9fc83253",
"metadata": {
"id": "get_batch_prediction:mbsdk,tst"
},
@@ -1181,7 +1130,6 @@
},
{
"cell_type": "markdown",
"id": "19466786",
"metadata": {
"id": "make_online_predictions:migration"
},
@@ -1191,7 +1139,6 @@
},
{
"cell_type": "markdown",
"id": "e97f1e55",
"metadata": {
"id": "deploy_model:migration,new,mbsdk"
},
@@ -1201,7 +1148,6 @@
},
{
"cell_type": "markdown",
"id": "d2745f77",
"metadata": {
"id": "deploy_model:mbsdk,automatic"
},
@@ -1214,7 +1160,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "6d30aa15",
"metadata": {
"id": "deploy_model:mbsdk,automatic"
},
@@ -1225,7 +1170,6 @@
},
{
"cell_type": "markdown",
"id": "c2c876d0",
"metadata": {
"id": "deploy_model:mbsdk,automatic"
},
@@ -1244,7 +1188,6 @@
},
{
"cell_type": "markdown",
"id": "9bb982a8",
"metadata": {
"id": "endpoints_predict:migration,new,mbsdk"
},
@@ -1254,20 +1197,18 @@
},
{
"cell_type": "markdown",
"id": "246945bb",
"metadata": {
"id": "get_test_item"
},
"source": [
"### Get test item\n",
"\n",
"You use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction."
"You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e21c3f76",
"metadata": {
"id": "get_test_item:automl,tst,csv"
},
@@ -1284,7 +1225,6 @@
},
{
"cell_type": "markdown",
"id": "95ffe1ea",
"metadata": {
"id": "predict_request:mbsdk,tst"
},
@@ -1313,7 +1253,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "16b7ab95",
"metadata": {
"id": "predict_request:mbsdk,tst"
},
@@ -1327,7 +1266,6 @@
},
{
"cell_type": "markdown",
"id": "f4c79c7f",
"metadata": {
"id": "predict_request:mbsdk,tst"
},
@@ -1339,7 +1277,6 @@
},
{
"cell_type": "markdown",
"id": "52717fb9",
"metadata": {
"id": "undeploy_model:mbsdk"
},
@@ -1352,7 +1289,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "7c164d13",
"metadata": {
"id": "undeploy_model:mbsdk"
},
@@ -1363,7 +1299,6 @@
},
{
"cell_type": "markdown",
"id": "4b844c87",
"metadata": {
"id": "cleanup:mbsdk"
},
@@ -1389,7 +1324,6 @@
{
"cell_type": "code",
"execution_count": null,
"id": "2ea906d0",
"metadata": {
"id": "cleanup:mbsdk"
},
@@ -1448,7 +1382,7 @@
" print(e)\n",
"\n",
" if \"BUCKET_NAME\" in globals():\n",
" ! gsutil rm -r $BUCKET_URI"
" ! gsutil rm -r $BUCKET_NAME"
]
}
],
@@ -117,15 +117,64 @@
"to generate a cost estimate based on your projected usage."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ze4-nDLfK4pw"
},
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Vertex AI Workbench**, your environment already meets\n",
"all the requirements to run this notebook. You can skip this step."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "gCuSR8GkAgzl"
},
"source": [
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
"You need the following:\n",
"\n",
"* The Google Cloud SDK\n",
"* Git\n",
"* Python 3\n",
"* virtualenv\n",
"* Jupyter notebook running in a virtual environment with Python 3\n",
"\n",
"The Google Cloud guide to [Setting up a Python development\n",
"environment](https://cloud.google.com/python/setup) and the [Jupyter\n",
"installation guide](https://jupyter.org/install) provide detailed instructions\n",
"for meeting these requirements. The following steps provide a condensed set of\n",
"instructions:\n",
"\n",
"1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n",
"\n",
"1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n",
"\n",
"1. [Install\n",
" virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n",
" and create a virtual environment that uses Python 3. Activate the virtual environment.\n",
"\n",
"1. To install Jupyter, run `pip install jupyter` on the\n",
"command-line in a terminal shell.\n",
"\n",
"1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n",
"\n",
"1. Open this notebook in the Jupyter Notebook Dashboard."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "i7EUnXsZhAGF"
},
"source": [
"### Installation\n",
"### Install additional packages\n",
"\n",
"Install the packages required for executing this notebook."
"Install additional package dependencies not installed in your notebook environment."
]
},
{
@@ -136,9 +185,23 @@
},
"outputs": [],
"source": [
"! pip3 install --upgrade tensorflow \\\n",
" google-cloud-aiplatform \\\n",
" scikit-learn -q"
"import os\n",
"\n",
"# The Vertex AI Workbench Notebook product has specific requirements\n",
"IS_WORKBENCH_NOTEBOOK = os.getenv(\"DL_ANACONDA_HOME\")\n",
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"\n",
"! pip3 install -U tensorflow $USER_FLAG\n",
"! python3 -m pip3 install {USER_FLAG} google-cloud-aiplatform --upgrade\n",
"! pip3 install scikit-learn {USER_FLAG}"
]
},
{
@@ -147,7 +210,9 @@
"id": "hhq5zEbGg0XX"
},
"source": [
"### Colab only: Uncomment the following cell to restart the kernel."
"### Restart the kernel\n",
"\n",
"After you install the additional packages, you need to restart the notebook kernel so it can find the packages."
]
},
{
@@ -158,11 +223,15 @@
},
"outputs": [],
"source": [
"# Automatically restart kernel after installs so that your environment can access the new packages\n",
"# import IPython\n",
"# Automatically restart kernel after installs\n",
"import os\n",
"\n",
"# app = IPython.Application.instance()\n",
"# app.kernel.do_shutdown(True)"
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
" import IPython\n",
"\n",
" app = IPython.Application.instance()\n",
" app.kernel.do_shutdown(True)"
]
},
{
@@ -171,21 +240,46 @@
"id": "lWEdiXsJg0XY"
},
"source": [
"## Before you begin"
"## Before you begin\n",
"\n",
"### Select a GPU runtime\n",
"\n",
"**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "8bc8a29f9001"
"id": "BF1j6f9HApxa"
},
"source": [
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
"\n",
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
"\n",
"1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n",
"\n",
"1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
"\n",
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "WReHDGG5g0XY"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, try the following:\n",
"* Run `gcloud config list`.\n",
"* Run `gcloud projects list`.\n",
"* See the support page: [Locate the project ID](https://support.google.com/googleapi/answer/7014113)"
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
@@ -196,10 +290,33 @@
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}\n",
"\n",
"# Set the project id\n",
"! gcloud config set project {PROJECT_ID}"
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "oM1iC_MfAts1"
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "TL9QIaVd9hvm"
},
"outputs": [],
"source": [
"!gcloud config set project $PROJECT_ID"
]
},
{
@@ -210,7 +327,16 @@
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable used by Vertex AI. Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
]
},
{
@@ -221,7 +347,41 @@
},
"outputs": [],
"source": [
"REGION = \"us-central1\" # @param {type: \"string\"}"
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "06571eb4063b"
},
"source": [
"#### UUID\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "697568e92bd6"
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"\n",
"\n",
"# Generate a uuid of length 8\n",
"def generate_uuid():\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=8))\n",
"\n",
"\n",
"UUID = generate_uuid()"
]
},
{
@@ -232,7 +392,8 @@
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"Depending on your Jupyter environment, you may have to manually authenticate. Follow the relevant instructions below."
"**If you are using Vertex AI Workbench**, your environment is already\n",
"authenticated. Skip this step."
]
},
{
@@ -241,37 +402,28 @@
"id": "sBCra4QMA2wR"
},
"source": [
"**1. Vertex AI Workbench**\n",
"* Do nothing as you are already authenticated."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ad1138a125ea"
},
"source": [
"**2. Local JupyterLab instance, uncomment and run:**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "ce6043da7b33"
},
"outputs": [],
"source": [
"# ! gcloud auth login"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "0367eac06a10"
},
"source": [
"**3. Colab, uncomment and run:**"
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"\n",
"2. Click **Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
]
},
{
@@ -282,18 +434,26 @@
},
"outputs": [],
"source": [
"# from google.colab import auth\n",
"# auth.authenticate_user()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "c13224697bfb"
},
"source": [
"**4. Service account or other**\n",
"* See how to grant Cloud Storage permissions to your service account at https://cloud.google.com/storage/docs/gsutil/commands/iam#ch-examples."
"import os\n",
"import sys\n",
"\n",
"# If you are running this notebook in Colab, run this cell and follow the\n",
"# instructions to authenticate your GCP account. This provides access to your\n",
"# Cloud Storage bucket and lets you submit training jobs and prediction\n",
"# requests.\n",
"\n",
"# If on Google Cloud Notebooks, then don't execute this code\n",
"if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n",
" if \"google.colab\" in sys.modules:\n",
" from google.colab import auth as google_auth\n",
"\n",
" google_auth.authenticate_user()\n",
"\n",
" # If you are running this notebook locally, replace the string below with the\n",
" # path to your service account key and run this cell to authenticate your GCP\n",
" # account.\n",
" elif not os.getenv(\"IS_TESTING\"):\n",
" %env GOOGLE_APPLICATION_CREDENTIALS ''"
]
},
{
@@ -304,7 +464,18 @@
"source": [
"### Create a Cloud Storage bucket\n",
"\n",
"Create a storage bucket to store intermediate artifacts such as datasets."
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"\n",
"When you submit a training job using the Vertex AI SDK, you upload a Python package\n",
"containing your training code to a Cloud Storage bucket. Vertex AI runs\n",
"the code from this package. In this tutorial, Vertex AI also saves the\n",
"trained model that results from your job in the same bucket. Using this model artifact, you can then\n",
"create Vertex AI model and endpoint resources in order to serve\n",
"online predictions.\n",
"\n",
"Set the name of your Cloud Storage bucket below. It must be unique across all\n",
"Cloud Storage buckets."
]
},
{
@@ -315,7 +486,21 @@
},
"outputs": [],
"source": [
"BUCKET_URI = \"gs://your-bucket-name-unique\" # @param {type:\"string\"}"
"BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}\n",
"BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cf221059d072"
},
"outputs": [],
"source": [
"if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n",
" BUCKET_NAME = PROJECT_ID + \"aip-\" + UUID\n",
" BUCKET_URI = f\"gs://{BUCKET_NAME}\""
]
},
{
@@ -335,7 +520,27 @@
},
"outputs": [],
"source": [
"! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI"
"! gsutil mb -l $REGION $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ucvCsknMCims"
},
"source": [
"Finally, validate access to your Cloud Storage bucket by examining its contents:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "vhOb7YnwClBb"
},
"outputs": [],
"source": [
"! gsutil ls -al $BUCKET_URI"
]
},
{
@@ -364,8 +569,6 @@
},
"outputs": [],
"source": [
"import os\n",
"\n",
"import pandas as pd\n",
"from google.cloud import aiplatform\n",
"from sklearn.metrics import mean_absolute_error, mean_squared_error\n",
@@ -398,7 +601,28 @@
},
"outputs": [],
"source": [
"EXPERIMENT_NAME = \"my-experiment-unique\""
"EXPERIMENT_NAME = \"\" # @param {type:\"string\"}"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "jWQLXXNVN4Lv"
},
"source": [
"If EXEPERIMENT_NAME is not set, set a default one below:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "Q1QInYWOKsmo"
},
"outputs": [],
"source": [
"if EXPERIMENT_NAME == \"\" or EXPERIMENT_NAME is None:\n",
" EXPERIMENT_NAME = \"my-experiment-\" + UUID"
]
},
{
@@ -588,9 +812,7 @@
},
"outputs": [],
"source": [
"aiplatform.start_run(\n",
" \"custom-training-run-unique\"\n",
") # Change this to your desired run name\n",
"aiplatform.start_run(\"custom-training-run-1\") # Change this to your desired run name\n",
"parameters = {\"epochs\": 10, \"num_units\": 64}\n",
"aiplatform.log_params(parameters)\n",
"\n",
@@ -827,12 +1049,6 @@
"# Delete dataset\n",
"ds.delete()\n",
"\n",
"# Delete experiment\n",
"experiment = aiplatform.Experiment(\n",
" experiment_name=EXPERIMENT_NAME, project=PROJECT_ID, location=REGION\n",
")\n",
"experiment.delete()\n",
"\n",
"# Delete the training job\n",
"job.delete()\n",
"\n",
File diff suppressed because it is too large Load Diff
@@ -29,7 +29,7 @@
"id": "JAPoU8Sm5E6e"
},
"source": [
"# Vertex AI Pipelines: Evaluating batch prediction results from AutoML Tabular regression model\n",
"# Vertex AI Pipelines: Evaluating BatchPrediction results from AutoML Tabular regression model\n",
"\n",
"<table align=\"left\">\n",
"\n",
@@ -76,11 +76,12 @@
"\n",
"This tutorial uses the following Google Cloud ML services and resources:\n",
"\n",
"- Vertex AI Datasets (Tabular)\n",
"- Vertex AI Training (AutoML Tabular Training)\n",
"- Vertex AI Batch predictions\n",
"- Vertex AI Pipelines\n",
"- Vertex AI Model Registry\n",
"- Vertex AI `AutoML`\n",
"- Vertex AI `TabularDataset` (AutoML)\n",
"- Vertex AI `AutoMLTabularTrainingJob`\n",
"- Vertex AI `BatchPrediction`\n",
"- Vertex AI `Pipeline`\n",
"- Vertex AI `Model Registry`\n",
"\n",
"\n",
"The steps performed include:\n",
@@ -89,9 +90,9 @@
"- Configure a `AutoMLTabularTrainingJob`\n",
"- Run the `AutoMLTabularTrainingJob` which returns a model\n",
"- Import a pre-trained `AutoML model resource` into the pipeline\n",
"- Run a `batch prediction` job in the pipeline\n",
"- Run a `batch prediction` job\n",
"- Evaulate the AutoML model using the `regression evaluation component`\n",
"- Import the Regression Metrics to the AutoML model resource"
"- Import the Classification Metrics to the AutoML model resource"
]
},
{
@@ -209,15 +210,16 @@
"IS_USER_MANAGED_WORKBENCH_NOTEBOOK = os.path.exists(\n",
" \"/opt/deeplearning/metadata/env_version\"\n",
")\n",
"\n",
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install --upgrade --quiet {USER_FLAG} google-cloud-aiplatform \\\n",
" google-cloud-pipeline-components==1.0.26 \\\n",
" kfp \\\n",
" matplotlib "
"! pip3 install --upgrade google-cloud-aiplatform {USER_FLAG} -q\n",
"! pip3 install google-cloud-pipeline-components==1.0.20 {USER_FLAG} -q\n",
"! pip3 install --upgrade kfp {USER_FLAG} -q\n",
"! pip3 install --upgrade matplotlib {USER_FLAG} -q"
]
},
{
@@ -706,7 +708,7 @@
"\n",
"Train a simple regression model using the created dataset using `Age` as the target column. \n",
"\n",
"**Set a display name and create the `AutoMLTabularTrainingJob` with appropriate data types specified for column transformations.**"
"##### Set a display name and create the `AutoMLTabularTrainingJob` with appropriate data types specified for column transformations."
]
},
{
@@ -737,25 +739,32 @@
" TRAINING_JOB_DISPLAY_NAME = \"train-pet-agefinder-automl_\" + UUID"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "1009f66873f5"
},
"source": [
"#### Define AutoMLTabularTrainingJob"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "6cb41277f4f3"
},
"source": [
"### Define AutoML Tabular training job\n",
"\n",
"An AutoML training job is created with the `AutoMLTabularTrainingJob` class, with the following parameters:\n",
"\n",
"- `display_name`: The human readable name for the `TrainingJob` resource.\n",
"- `optimization_prediction_type`: The type of prediction the AutoML Model is to produce. Ex: regression, classification.\n",
"- `column_transformations`: Transformations to apply to the input columns (i.e. columns other than the targetColumn). Each transformation may produce multiple result values from the column's value, and all are used for training. \n",
"- `optimization_prediction_type`: The type of prediction the AutoML Model is to produce. Ex: regression,classification\n",
"- `column_transformations`: Transformations to apply to the input columns (i.e. columns other than the targetColumn). Each transformation may produce multiple result values from the column's value, and all are used for training. When creating transformation for BigQuery Struct column, the column should be flattened using \".\" as the delimiter. Only columns with no child should have a transformation. If an input column has no transformations on it, such a column is ignored by the training, except for the targetColumn, which should have no transformations defined on. Only one of column_transformations or column_specs should be passed. Consider using column_specs as column_transformations will be deprecated eventually. If none of column_transformations or column_specs is passed, the local credentials being used will try setting column_transformations to \"auto\". To do this, the local credentials require read access to the Cloud Storage or BigQuery training data source.\n",
"- `optimization_objective`: The optimization objective to minimize or maximize.\n",
" - `minimize-rmse`\n",
" - `minimize-mae`\n",
" - `minimize-rmsle`\n",
"\n",
"Learn about [AutoMLTabularTrainingJob](https://cloud.google.com/python/docs/reference/aiplatform/latest/google.cloud.aiplatform.AutoMLTabularTrainingJob) "
"To learn more about `AutoMLTabularTrainingJob` click [here](https://cloud.google.com/python/docs/reference/aiplatform/latest/google.cloud.aiplatform.AutoMLTabularTrainingJob) "
]
},
{
@@ -796,7 +805,7 @@
"id": "391c51c98647"
},
"source": [
"#### Set the display name for the model."
"##### Set the display name for the model."
]
},
{
@@ -827,14 +836,21 @@
" MODEL_DISPLAY_NAME = \"pet-agefinder-prediction-model_\" + UUID"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "97c335a53595"
},
"source": [
"#### Run the training job"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "de7e24205889"
},
"source": [
"#### Run the training job\n",
"\n",
"Next, you start the training job by invoking the method `run`, with the following parameters:\n",
"\n",
"- `dataset`: The `Dataset` resource to train the model.\n",
@@ -843,9 +859,10 @@
"- `validation_fraction_split`: The percentage of the dataset to use for validation.\n",
"- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n",
"- `model_display_name`: The human readable name for the trained model.\n",
"- `budget_milli_node_hours`: The train budget of creating this Model, expressed in milli node hours i.e. 1,000 value in this field means 1 node hour. \n",
"- `disable_early_stopping`: If true, the entire budget is used.\n",
"- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n",
"\n",
"The training job takes roughly 3 hours to finish."
"The training job takes roughly 1.5-2 hours to finish."
]
},
{
@@ -864,6 +881,7 @@
" validation_fraction_split=0.1,\n",
" test_fraction_split=0.1,\n",
" model_display_name=MODEL_DISPLAY_NAME,\n",
" disable_early_stopping=False,\n",
" budget_milli_node_hours=1000,\n",
")"
]
@@ -914,44 +932,14 @@
{
"cell_type": "markdown",
"metadata": {
"id": "2241f3739e03"
"id": "581a188f0453"
},
"source": [
"## Create Pipeline for evaluations\n",
"\n",
"Now, you run a Vertex AI BatchPrediction job and generate evaluations and feature-attributions on its results by creating a Vertex AI pipeline using the components available from the [google-cloud-pipeline-components](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/index.html) python package. \n",
"Now, you run a Vertex AI BatchPrediction job and generate evaluations and feature-attributions on its results. \n",
"\n",
"**Set a display name for your pipeline.**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "d17a0268020f"
},
"outputs": [],
"source": [
"PIPELINE_DISPLAY_NAME = \"[your-pipeline-display-name]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "9e41d48d1f48"
},
"outputs": [],
"source": [
"# If no display name is set, use the default one\n",
"if (\n",
" PIPELINE_DISPLAY_NAME == \"[your-pipeline-display-name]\"\n",
" or PIPELINE_DISPLAY_NAME == \"\"\n",
" or PIPELINE_DISPLAY_NAME is None\n",
"):\n",
" PIPELINE_DISPLAY_NAME = (\n",
" f\"vertex-evaluation-automl-tabular-regression-feature-attribution-{UUID}\"\n",
" )"
"To do so, you create a Vertex AI pipeline using the components available from the [`google-cloud-pipeline-components`](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/index.html) python package.\n"
]
},
{
@@ -966,14 +954,21 @@
"\n",
"The pipeline uses the following components:\n",
"\n",
"- `GetVertexModelOp`: Gets a Vertex AI Model Artifact. \n",
"- `EvaluationDataSamplerOp`: Randomly downsamples an input dataset to a specified size for computing Vertex XAI feature attributions for AutoML Tables and custom models. Creates a Dataflow job with Apache Beam to downsample the dataset.\n",
"- `ModelBatchPredictOp`: Creates a Google Cloud Vertex BatchPredictionJob and waits for it to complete.\n",
"- `ModelEvaluationRegressionOp`: Compute evaluation metrics on a trained model’s batch prediction results. Creates a Dataflow job with Apache Beam and TFMA to compute evaluation metrics. Supports regression for tabular data. \n",
"- `ModelEvaluationFeatureAttributionOp`: Compute feature attribution on a trained model’s batch explanation results. Creates a Dataflow job with Apache Beam and TFMA to compute feature attributions.\n",
"- `ModelImportEvaluationOp`: Imports a model evaluation artifact to an existing Vertex AI Model with ModelService.ImportModelEvaluation.\n",
"\n",
"Learn more about [Google Cloud Pipeline Evaluation Components](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.26/google_cloud_pipeline_components.experimental.evaluation.html)"
"- `GetVertexModelOp`: Gets a Vertex Model Artifact. For more details, please check [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/google_cloud_pipeline_components.experimental.evaluation.html#google_cloud_pipeline_components.experimental.evaluation.GetVertexModelOp).\n",
"- `EvaluationDataSamplerOp`: Randomly downsamples an input dataset to a specified size for computing Vertex XAI feature attributions for AutoML Tables and custom models. Creates a Dataflow job with Apache Beam to downsample the dataset. For more details, please check [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/google_cloud_pipeline_components.experimental.evaluation.html#google_cloud_pipeline_components.experimental.evaluation.EvaluationDataSamplerOp).\n",
"- `ModelBatchPredictOp`: Creates a Google Cloud Vertex BatchPredictionJob and waits for it to complete. For more details, please check [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/google_cloud_pipeline_components.aiplatform.html#google_cloud_pipeline_components.aiplatform.ModelBatchPredictOp).\n",
"- `ModelEvaluationRegressionOp`: Compute evaluation metrics on a trained model’s batch prediction results. Creates a Dataflow job with Apache Beam and TFMA to compute evaluation metrics. Supports regression for tabular data.[here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/google_cloud_pipeline_components.experimental.evaluation.html#google_cloud_pipeline_components.experimental.evaluation.ModelEvaluationRegressionOp).\n",
"- `ModelEvaluationFeatureAttributionOp`: Compute feature attribution on a trained model’s batch explanation results. Creates a Dataflow job with Apache Beam and TFMA to compute feature attributions. For more details, please check [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/google_cloud_pipeline_components.experimental.evaluation.html#google_cloud_pipeline_components.experimental.evaluation.ModelEvaluationFeatureAttributionOp).\n",
"- `ModelImportEvaluationOp`: Imports a model evaluation artifact to an existing Vertex model with ModelService.ImportModelEvaluation. For more details, please check [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.17/google_cloud_pipeline_components.experimental.evaluation.html#google_cloud_pipeline_components.experimental.evaluation.ModelImportEvaluationOp)."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "ce6beLsXASnK"
},
"source": [
"## Model Evaluation"
]
},
{
@@ -984,7 +979,9 @@
},
"outputs": [],
"source": [
"@kfp.dsl.pipeline(name=PIPELINE_DISPLAY_NAME)\n",
"@kfp.dsl.pipeline(\n",
" name=\"vertex-evaluation-automl-tabular-regression-feature-attribution\"\n",
")\n",
"def evaluation_automl_tabular_feature_attribution_pipeline(\n",
" project: str,\n",
" location: str,\n",
@@ -993,9 +990,14 @@
" target_column_name: str,\n",
" batch_predict_gcs_source_uris: list,\n",
" batch_predict_instances_format: str,\n",
" batch_predict_sample_size: int,\n",
" batch_predict_predictions_format: str = \"jsonl\",\n",
" batch_predict_machine_type: str = \"n1-standard-4\",\n",
" batch_predict_explanation_metadata: dict = {},\n",
" batch_predict_explanation_parameters: dict = {},\n",
" batch_predict_explanation_data_sample_size: int = 10000,\n",
" dataflow_max_num_workers: int = 5,\n",
" dataflow_use_public_ips: bool = True,\n",
" encryption_spec_key_name: str = \"\",\n",
"):\n",
"\n",
" from google_cloud_pipeline_components.aiplatform import ModelBatchPredictOp\n",
@@ -1014,7 +1016,7 @@
" root_dir=root_dir,\n",
" gcs_source_uris=batch_predict_gcs_source_uris,\n",
" instances_format=batch_predict_instances_format,\n",
" sample_size=batch_predict_sample_size,\n",
" sample_size=batch_predict_explanation_data_sample_size,\n",
" )\n",
"\n",
" # Run Batch Explanations\n",
@@ -1028,20 +1030,25 @@
" predictions_format=batch_predict_predictions_format,\n",
" gcs_destination_output_uri_prefix=root_dir,\n",
" machine_type=batch_predict_machine_type,\n",
" encryption_spec_key_name=encryption_spec_key_name,\n",
" # Set the explanation parameters\n",
" generate_explanation=True,\n",
" explanation_parameters=batch_predict_explanation_parameters,\n",
" explanation_metadata=batch_predict_explanation_metadata,\n",
" )\n",
"\n",
" # Run evaluation based on prediction type and feature attribution component.\n",
" # After, import the model evaluations to the Vertex AI model.\n",
" # After, import the model evaluations to the Vertex model.\n",
" eval_task = ModelEvaluationRegressionOp(\n",
" project=project,\n",
" location=location,\n",
" root_dir=root_dir,\n",
" target_field_name=target_column_name,\n",
" ground_truth_column=target_column_name,\n",
" predictions_gcs_source=batch_explain_task.outputs[\"gcs_output_directory\"],\n",
" predictions_format=batch_predict_predictions_format,\n",
" prediction_score_column=\"prediction.value\",\n",
" dataflow_max_workers_num=dataflow_max_num_workers,\n",
" dataflow_use_public_ips=dataflow_use_public_ips,\n",
" encryption_spec_key_name=encryption_spec_key_name,\n",
" )\n",
"\n",
" # Get Feature Attributions\n",
@@ -1051,6 +1058,9 @@
" root_dir=root_dir,\n",
" predictions_format=\"jsonl\",\n",
" predictions_gcs_source=batch_explain_task.outputs[\"gcs_output_directory\"],\n",
" dataflow_max_workers_num=dataflow_max_num_workers,\n",
" dataflow_use_public_ips=dataflow_use_public_ips,\n",
" encryption_spec_key_name=encryption_spec_key_name,\n",
" )\n",
"\n",
" ModelImportEvaluationOp(\n",
@@ -1094,7 +1104,37 @@
"source": [
"### Define the parameters to run the pipeline\n",
"\n",
"Specify the required parameters to run the pipeline.\n"
"Specify the required parameters to run the pipeline.\n",
"\n",
"Set a display name for your pipeline."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "8f17c5c7b3e3"
},
"outputs": [],
"source": [
"PIPELINE_DISPLAY_NAME = \"[your-pipeline-display-name]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "1aa7d7bbb1c9"
},
"outputs": [],
"source": [
"# If no display name is set, use the default one\n",
"if (\n",
" PIPELINE_DISPLAY_NAME == \"[your-pipeline-display-name]\"\n",
" or PIPELINE_DISPLAY_NAME == \"\"\n",
" or PIPELINE_DISPLAY_NAME is None\n",
"):\n",
" PIPELINE_DISPLAY_NAME = \"pet_agefinder_\" + UUID"
]
},
{
@@ -1103,7 +1143,7 @@
"id": "90f424d5dca0"
},
"source": [
"To pass the required arguments to the pipeline, you define the following parameters below:\n",
"To pass the required arguments to the pipeline, you define the following paramters below:\n",
"\n",
"- `project`: Project ID.\n",
"- `location`: Region where the pipeline is run.\n",
@@ -1112,7 +1152,7 @@
"- `target_column_name`: Name of the column to be used as the target for regression.\n",
"- `batch_predict_gcs_source_uris`: List of the Cloud Storage bucket uris of input instances for batch prediction.\n",
"- `batch_predict_instances_format`: Format of the input instances for batch prediction. Can be \"jsonl\", \"csv\" or \"bigquery\".\n",
"- `batch_predict_sample_size`: Size of the samples to be considered for batch prediction and evaluation."
"- `batch_predict_explanation_data_sample_size`: Size of the samples to be considered for batch prediction and evaluation."
]
},
{
@@ -1124,7 +1164,6 @@
"outputs": [],
"source": [
"PIPELINE_ROOT = f\"{BUCKET_URI}/pipeline_root/pet_agefinder_{UUID}\"\n",
"batch_predict_sample_size = 10000\n",
"parameters = {\n",
" \"project\": PROJECT_ID,\n",
" \"location\": REGION,\n",
@@ -1133,7 +1172,7 @@
" \"target_column_name\": \"Age\",\n",
" \"batch_predict_gcs_source_uris\": [DATA_SOURCE],\n",
" \"batch_predict_instances_format\": \"csv\",\n",
" \"batch_predict_sample_size\": batch_predict_sample_size,\n",
" \"batch_predict_explanation_data_sample_size\": 3000,\n",
"}"
]
},
@@ -1146,7 +1185,7 @@
"Next, you create the pipeline job, with the following parameters:\n",
"\n",
"- `display_name`: The user-defined name of this Pipeline.\n",
"- `template_path`: The path of PipelineJob or PipelineSpec JSON or YAML file. It can be a local path, a Google Cloud Storage URI, or an Artifact Registry URI.\n",
"- `template_path`: The path of PipelineJob or PipelineSpec JSON or YAML file. It can be a local path, a Google Cloud Storage URI (e.g. \"gs://project.name\"), or an Artifact Registry URI (e.g. \"https://us-central1-kfp.pkg.dev/proj/repo/pack/latest\").\n",
"- `parameter_values`: The mapping from runtime parameter names to its values that control the pipeline run.\n",
"- `enable_caching`: Whether to turn on caching for the run. If this is not set, defaults to the compile time settings, which are True for all tasks by default, while users may specify different caching options for individual tasks. If this is set, the setting applies to all tasks in the pipeline. Overrides the compile time settings.\n"
]
@@ -1157,9 +1196,7 @@
"id": "e8dce0638349"
},
"source": [
"Run the pipeline using the configured `SERVICE_ACCOUNT`.\n",
"\n",
"**The pipeline takes about 2 hours to complete.**\n"
"Run the pipeline using the configured `SERVICE_ACCOUNT`."
]
},
{
@@ -1183,12 +1220,29 @@
{
"cell_type": "markdown",
"metadata": {
"id": "U2zocUvk2YVs"
"id": "625960707c60"
},
"source": [
"## Model Evaluation"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "mKRTDi8ioXBY"
},
"source": [
"In the results from last step, click on the generated link to see your run in the Cloud Console.\n",
"\n",
"In the UI, many of the pipeline DAG nodes will expand or collapse when you click on them. Here is a partially-expanded view of the DAG (click image to see larger version).\n",
"In the UI, many of the pipeline DAG nodes will expand or collapse when you click on them. Here is a partially-expanded view of the DAG (click image to see larger version).\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "U2zocUvk2YVs"
},
"source": [
"<img src=\"images/automl_tabular_regression_evaluation_pipeline.PNG\" style=\"height:622px;width:726px\"></img>"
]
},
@@ -1240,15 +1294,6 @@
"### Visualize the metrics\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "14MBD57k0Fng"
},
"source": [
"After the evalution pipeline is finished, run the below cell to visualize the evaluation metrics."
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -1283,7 +1328,7 @@
"\n",
"Feature attributions indicate how much each feature in your model contributed to the predictions for each given instance.\n",
"\n",
"Learn more about [Feature Attributions](https://cloud.google.com/vertex-ai/docs/explainable-ai/overview#feature_attributions)\n",
"To learn more about Feature Attributions click [here](https://cloud.google.com/vertex-ai/docs/explainable-ai/overview#feature_attributions)\n",
"\n",
"Run the below cell to get the feature attributions. "
]
@@ -807,7 +807,6 @@
" training_fraction_split=0.8,\n",
" test_fraction_split=0.2,\n",
")\n",
"\n",
"print(model)"
]
},
@@ -1038,7 +1037,7 @@
" model_name: str,\n",
" target_column_name: str,\n",
" ground_truth_gcs_uri: list,\n",
" class_labels: list = \"{}\",\n",
" key_columns: list,\n",
" batch_predict_instances_format: str = \"jsonl\",\n",
" batch_predict_predictions_format: str = \"jsonl\",\n",
" batch_predict_machine_type: str = \"n1-standard-16\",\n",
@@ -1049,8 +1048,8 @@
" from google_cloud_pipeline_components.aiplatform import ModelBatchPredictOp\n",
" from google_cloud_pipeline_components.experimental import evaluation\n",
" from google_cloud_pipeline_components.experimental.evaluation import (\n",
" EvaluationDataSamplerOp, ModelEvaluationClassificationOp,\n",
" ModelImportEvaluationOp, TargetFieldDataRemoverOp)\n",
" EvaluationDataSamplerOp, EvaluationDataSplitterOp,\n",
" ModelEvaluationClassificationOp, ModelImportEvaluationOp)\n",
"\n",
" get_model_task = evaluation.GetVertexModelOp(model_resource_name=model_name)\n",
"\n",
@@ -1064,13 +1063,13 @@
" )\n",
"\n",
" # Run Data-splitter task\n",
" data_splitter_task = TargetFieldDataRemoverOp(\n",
" data_splitter_task = EvaluationDataSplitterOp(\n",
" project=project,\n",
" location=location,\n",
" root_dir=root_dir,\n",
" gcs_source_uris=data_sampler_task.outputs[\"gcs_output_directory\"],\n",
" instances_format=batch_predict_instances_format,\n",
" target_field_name=target_column_name,\n",
" ground_truth_column=target_column_name,\n",
" )\n",
"\n",
" # Run Batch Prediction.\n",
@@ -1094,12 +1093,9 @@
" project=project,\n",
" location=location,\n",
" root_dir=root_dir,\n",
" key_columns=key_columns,\n",
" ground_truth_gcs_source=data_sampler_task.outputs[\"gcs_output_directory\"],\n",
" target_field_name=target_column_name,\n",
" prediction_score_column=\"prediction.confidence\",\n",
" prediction_label_column=\"prediction.displayName\",\n",
" class_labels=[\"brush_hair\", \"cartwheel\"],\n",
" ground_truth_format=batch_predict_instances_format,\n",
" ground_truth_column=target_column_name,\n",
" predictions_format=batch_predict_predictions_format,\n",
" predictions_gcs_source=batch_predict_task.outputs[\"gcs_output_directory\"],\n",
" )\n",
@@ -1176,6 +1172,7 @@
" \"model_name\": MODEL_RSC_NAME,\n",
" \"target_column_name\": LABEL_COLUMN,\n",
" \"ground_truth_gcs_uri\": [gcs_ground_truth_uri],\n",
" \"key_columns\": [\"content\", \"mimeType\", \"timeSegmentStart\", \"timeSegmentEnd\"],\n",
" \"batch_predict_instances_format\": \"jsonl\",\n",
" \"batch_predict_sample_size\": SAMPLE_SIZE,\n",
"}"
@@ -1669,9 +1669,9 @@
"\n",
" from google_cloud_pipeline_components.aiplatform import ModelBatchPredictOp\n",
" from google_cloud_pipeline_components.experimental.evaluation import (\n",
" EvaluationDataSamplerOp, GetVertexModelOp,\n",
" EvaluationDataSamplerOp, EvaluationDataSplitterOp, GetVertexModelOp,\n",
" ModelEvaluationFeatureAttributionOp, ModelEvaluationRegressionOp,\n",
" ModelImportEvaluationOp, TargetFieldDataRemoverOp)\n",
" ModelImportEvaluationOp)\n",
"\n",
" # Get the Vertex AI model resource\n",
" get_model_task = GetVertexModelOp(model_resource_name=model_name)\n",
@@ -1687,13 +1687,13 @@
" )\n",
"\n",
" # Run Data-splitter task\n",
" data_splitter_task = TargetFieldDataRemoverOp(\n",
" data_splitter_task = EvaluationDataSplitterOp(\n",
" project=project,\n",
" location=location,\n",
" root_dir=root_dir,\n",
" gcs_source_uris=data_sampler_task.outputs[\"gcs_output_directory\"],\n",
" instances_format=batch_predict_instances_format,\n",
" target_field_name=target_column_name,\n",
" ground_truth_column=target_column_name,\n",
" )\n",
"\n",
" # Run Batch Explanations\n",
@@ -1720,9 +1720,10 @@
" predictions_gcs_source=batch_explain_task.outputs[\"gcs_output_directory\"],\n",
" ground_truth_format=\"jsonl\",\n",
" ground_truth_gcs_source=data_sampler_task.outputs[\"gcs_output_directory\"],\n",
" key_columns=key_columns,\n",
" predictions_format=batch_predict_predictions_format,\n",
" prediction_score_column=\"prediction\",\n",
" target_field_name=target_column_name,\n",
" ground_truth_column=target_column_name,\n",
" )\n",
"\n",
" # Get Feature Attributions\n",
Binary file not shown.

Before

Width:  |  Height:  |  Size: 109 KiB

After

Width:  |  Height:  |  Size: 55 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 90 KiB

File diff suppressed because it is too large Load Diff
@@ -229,6 +229,27 @@
" app.kernel.do_shutdown(True)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "check_versions"
},
"source": [
"Check the versions of the packages you installed. The KFP SDK version should be >=1.6."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "check_versions:kfp,gcpc"
},
"outputs": [],
"source": [
"! python3 -c \"import kfp; print('KFP SDK version: {}'.format(kfp.__version__))\"\n",
"! python3 -c \"import google_cloud_pipeline_components; print('google_cloud_pipeline_components version: {}'.format(google_cloud_pipeline_components.__version__))\""
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -259,17 +280,6 @@
"**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$`."
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "5aee4379e8e5"
},
"source": [
"#### Set your project ID\n",
"\n",
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -611,12 +621,8 @@
},
"outputs": [],
"source": [
"import os\n",
"from typing import Any, Dict, List\n",
"\n",
"import google.cloud.aiplatform as aip\n",
"import kfp\n",
"from kfp.v2 import compiler # noqa: F811"
"import kfp"
]
},
{
@@ -703,7 +709,7 @@
"def pipeline(\n",
" project: str = PROJECT_ID,\n",
" model_display_name: str = MODEL_DISPLAY_NAME,\n",
" serving_container_image_uri: str = \"us-docker.pkg.dev/vertex-ai/prediction/tf2-cpu.2-9:latest\",\n",
" serving_container_image_uri: str = \"us-docker.pkg.dev/cloud-aiplatform/prediction/tf2-cpu.2-3:latest\",\n",
"):\n",
" from google_cloud_pipeline_components.types import artifact_types\n",
" from google_cloud_pipeline_components.v1.custom_job import \\\n",
@@ -738,7 +744,7 @@
" artifact_class=artifact_types.UnmanagedContainerModel,\n",
" metadata={\n",
" \"containerSpec\": {\n",
" \"imageUri\": \"us-docker.pkg.dev/vertex-ai/prediction/tf2-cpu.2-9:latest\",\n",
" \"imageUri\": \"us-docker.pkg.dev/cloud-aiplatform/prediction/tf2-cpu.2-3:latest\",\n",
" },\n",
" },\n",
" ).after(custom_job_task)\n",
@@ -784,6 +790,8 @@
},
"outputs": [],
"source": [
"from kfp.v2 import compiler # noqa: F811\n",
"\n",
"compiler.Compiler().compile(\n",
" pipeline_func=pipeline,\n",
" package_path=\"tabular_regression_pipeline.json\",\n",
@@ -852,54 +860,15 @@
"project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n",
"\n",
"Otherwise, you can delete the individual resources you created in this tutorial -- *Note:* this is auto-generated and not all resources may be applicable for this tutorial:\n",
"### Get resources from the pipline to clean up\n",
"Function to get details of a task"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "50f5f0d96294"
},
"outputs": [],
"source": [
"def get_task_detail(\n",
" task_details: List[Dict[str, Any]], task_name: str\n",
") -> List[Dict[str, Any]]:\n",
" for task_detail in task_details:\n",
" if task_detail.task_name == task_name:\n",
" return task_detail"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "0966cd56fc3e"
},
"outputs": [],
"source": [
"pipeline_task_details = (\n",
" job.gca_resource.job_detail.task_details\n",
") # fetch pipeline task details\n",
"\n",
"\n",
"# fetch endpoint from pipeline and delete the endpoint\n",
"endpoint_task = get_task_detail(pipeline_task_details, \"endpoint-create\")\n",
"endpoint_resourceName = (\n",
" endpoint_task.outputs[\"endpoint\"].artifacts[0].metadata[\"resourceName\"]\n",
")\n",
"endpoint = aip.Endpoint(endpoint_resourceName)\n",
"# undeploy model from endpoint\n",
"endpoint.undeploy_all()\n",
"endpoint.delete()\n",
"\n",
"# fetch model from pipeline and delete the model\n",
"model_task = get_task_detail(pipeline_task_details, \"model-upload\")\n",
"model_resourceName = model_task.outputs[\"model\"].artifacts[0].metadata[\"resourceName\"]\n",
"model = aip.Model(model_resourceName)\n",
"model.delete()"
"- Dataset\n",
"- Pipeline\n",
"- Model\n",
"- Endpoint\n",
"- Batch Job\n",
"- Custom Job\n",
"- Hyperparameter Tuning Job\n",
"- Cloud Storage Bucket"
]
},
{
@@ -910,6 +879,93 @@
},
"outputs": [],
"source": [
"delete_dataset = True\n",
"delete_pipeline = True\n",
"delete_model = True\n",
"delete_endpoint = True\n",
"delete_batchjob = True\n",
"delete_customjob = True\n",
"delete_hptjob = True\n",
"\n",
"try:\n",
" if delete_model and \"DISPLAY_NAME\" in globals():\n",
" models = aip.Model.list(\n",
" filter=f\"display_name={DISPLAY_NAME}\", order_by=\"create_time\"\n",
" )\n",
" model = models[0]\n",
" aip.Model.delete(model)\n",
" print(\"Deleted model:\", model)\n",
"except Exception as e:\n",
" print(e)\n",
"\n",
"try:\n",
" if delete_endpoint and \"DISPLAY_NAME\" in globals():\n",
" endpoints = aip.Endpoint.list(\n",
" filter=f\"display_name={DISPLAY_NAME}_endpoint\", order_by=\"create_time\"\n",
" )\n",
" endpoint = endpoints[0]\n",
" endpoint.undeploy_all()\n",
" aip.Endpoint.delete(endpoint.resource_name)\n",
" print(\"Deleted endpoint:\", endpoint)\n",
"except Exception as e:\n",
" print(e)\n",
"\n",
"if delete_dataset and \"DISPLAY_NAME\" in globals():\n",
" if \"tabular\" == \"tabular\":\n",
" try:\n",
" datasets = aip.TabularDataset.list(\n",
" filter=f\"display_name={DISPLAY_NAME}\", order_by=\"create_time\"\n",
" )\n",
" dataset = datasets[0]\n",
" aip.TabularDataset.delete(dataset.resource_name)\n",
" print(\"Deleted dataset:\", dataset)\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" if \"tabular\" == \"image\":\n",
" try:\n",
" datasets = aip.ImageDataset.list(\n",
" filter=f\"display_name={DISPLAY_NAME}\", order_by=\"create_time\"\n",
" )\n",
" dataset = datasets[0]\n",
" aip.ImageDataset.delete(dataset.resource_name)\n",
" print(\"Deleted dataset:\", dataset)\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" if \"tabular\" == \"text\":\n",
" try:\n",
" datasets = aip.TextDataset.list(\n",
" filter=f\"display_name={DISPLAY_NAME}\", order_by=\"create_time\"\n",
" )\n",
" dataset = datasets[0]\n",
" aip.TextDataset.delete(dataset.resource_name)\n",
" print(\"Deleted dataset:\", dataset)\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
" if \"tabular\" == \"video\":\n",
" try:\n",
" datasets = aip.VideoDataset.list(\n",
" filter=f\"display_name={DISPLAY_NAME}\", order_by=\"create_time\"\n",
" )\n",
" dataset = datasets[0]\n",
" aip.VideoDataset.delete(dataset.resource_name)\n",
" print(\"Deleted dataset:\", dataset)\n",
" except Exception as e:\n",
" print(e)\n",
"\n",
"try:\n",
" if delete_pipeline and \"DISPLAY_NAME\" in globals():\n",
" pipelines = aip.PipelineJob.list(\n",
" filter=f\"display_name={DISPLAY_NAME}\", order_by=\"create_time\"\n",
" )\n",
" pipeline = pipelines[0]\n",
" aip.PipelineJob.delete(pipeline.resource_name)\n",
" print(\"Deleted pipeline:\", pipeline)\n",
"except Exception as e:\n",
" print(e)\n",
"\n",
"delete_bucket = False\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
" ! gsutil rm -r $BUCKET_URI"
@@ -77,7 +77,7 @@
"This tutorial uses the following Google Cloud ML services and resources:\n",
"\n",
"- Vertex AI Training\n",
"- Vertex AI Pipelines\n",
"- Vertex Pipelines\n",
"- Cloud Storage\n",
"\n",
"The steps performed include:\n",
@@ -97,7 +97,7 @@
"### Dataset\n",
"\n",
"The dataset you will be using is [Bank Marketing](https://archive.ics.uci.edu/ml/datasets/bank+marketing).\n",
"The data is for direct marketing campaigns (phone calls) of a Portuguese banking institution. The binary classification goal is to predict if a client subscribe a term deposit. For this notebook, you randomly selected 90% of the rows in the original dataset and saved them in a train.csv file hosted on Cloud Storage. To download the file, click [here](https://storage.googleapis.com/cloud-samples-data/vertex-ai/tabular-workflows/datasets/bank-marketing/train.csv)."
"The data is for direct marketing campaigns (phone calls) of a Portuguese banking institution. The binary classification goal is to predict if a client will subscribe a term deposit. For this notebook, we randomly selected 90% of the rows in the original dataset and saved them in a train.csv file hosted on Cloud Storage. To download the file, click [here](https://storage.googleapis.com/cloud-samples-data/vertex-ai/tabular-workflows/datasets/bank-marketing/train.csv)."
]
},
{
@@ -128,7 +128,7 @@
"source": [
"### Set up your local development environment\n",
"\n",
"**If you are using Colab or Vertex AI SDK Workbench Notebooks**, your environment already meets\n",
"**If you are using Colab or Vertex AI Workbench Notebooks**, your environment already meets\n",
"all the requirements to run this notebook. You can skip this step.\n",
"\n",
"**Otherwise**, make sure your environment meets this notebook's requirements.\n",
@@ -251,7 +251,7 @@
"\n",
"3. [Enable the following APIs: Vertex AI APIs, Dataflow APIs, Compute Engine APIs, and Cloud Storage.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,dataflow.googleapis.com,compute_component,storage-component.googleapis.com)\n",
"\n",
"4. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"4. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"5. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
@@ -438,7 +438,7 @@
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
"\n",
"All training related files (TF model checkpoint, TensorBoard file, etc) will be saved to the GCS bucket. The pipeline not clean up the files since some of them might be useful for you, **please make sure to clean up the files**. For easy cleanup, you can set [GCS bucket level TTL](https://cloud.google.com/storage/docs/lifecycle).\n",
"All training related files (TF model checkpoint, TensorBoard file, etc) will be saved to the GCS bucket. The pipeline will not clean up the files since some of them might be useful for you, **please make sure to clean up the files**. For easy cleanup, you can set [GCS bucket level TTL](https://cloud.google.com/storage/docs/lifecycle).\n",
"\n",
"Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization.\n"
]
@@ -512,77 +512,6 @@
"! gsutil ls -al $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "44accda192d5"
},
"source": [
"#### Service Account\n",
"\n",
"You use a service account to create Vertex AI Pipeline jobs. If you do not want to use your project's Compute Engine service account, set `SERVICE_ACCOUNT` to another service account ID."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "c65d12a97f45"
},
"outputs": [],
"source": [
"SERVICE_ACCOUNT = \"[your-service-account]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "604ae09ab6d3"
},
"outputs": [],
"source": [
"if (\n",
" SERVICE_ACCOUNT == \"\"\n",
" or SERVICE_ACCOUNT is None\n",
" or SERVICE_ACCOUNT == \"[your-service-account]\"\n",
"):\n",
" # Get your service account from gcloud\n",
" if not IS_COLAB:\n",
" shell_output = !gcloud auth list 2>/dev/null\n",
" SERVICE_ACCOUNT = shell_output[2].replace(\"*\", \"\").strip()\n",
"\n",
" else: # IS_COLAB:\n",
" shell_output = ! gcloud projects describe $PROJECT_ID\n",
" project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
" SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
"\n",
" print(\"Service Account:\", SERVICE_ACCOUNT)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "d1ecb60964d5"
},
"source": [
"#### Set service account access for Vertex AI Pipelines\n",
"Run the following commands to grant your service account access to read and write pipeline artifacts in the bucket that you created in the previous step. You only need to run this step once per service account."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "a592f0a380c2"
},
"outputs": [],
"source": [
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
"\n",
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
]
},
{
"cell_type": "markdown",
"metadata": {
@@ -648,13 +577,11 @@
},
"outputs": [],
"source": [
"# Get the mdoel artifacts path from task details.\n",
"def get_model_artifacts_path(task_details: List[Dict[str, Any]], task_name: str) -> str:\n",
" task = get_task_detail(task_details, task_name)\n",
" return task.outputs[\"unmanaged_container_model\"].artifacts[0].uri\n",
"\n",
"\n",
"# Get the model uri from the task details.\n",
"def get_model_uri(task_details: List[Dict[str, Any]]) -> str:\n",
" task = get_task_detail(task_details, \"model-upload\")\n",
" # in format https://<location>-aiplatform.googleapis.com/v1/projects/<project_number>/locations/<location>/models/<model_id>\n",
@@ -662,14 +589,12 @@
" return f\"https://console.cloud.google.com/vertex-ai/locations/{REGION}/models/{model_id}?project={PROJECT_ID}\"\n",
"\n",
"\n",
"# Get the bucket name and path.\n",
"def get_bucket_name_and_path(uri: str) -> str:\n",
" no_prefix_uri = uri[len(\"gs://\") :]\n",
" splits = no_prefix_uri.split(\"/\")\n",
" return splits[0], \"/\".join(splits[1:])\n",
"\n",
"\n",
"# Get the content from the bucket.\n",
"def download_from_gcs(uri: str) -> str:\n",
" bucket_name, path = get_bucket_name_and_path(uri)\n",
" storage_client = storage.Client(project=PROJECT_ID)\n",
@@ -678,7 +603,6 @@
" return blob.download_as_string()\n",
"\n",
"\n",
"# Upload content in to the bucket.\n",
"def write_to_gcs(uri: str, content: str):\n",
" bucket_name, path = get_bucket_name_and_path(uri)\n",
" storage_client = storage.Client()\n",
@@ -687,7 +611,6 @@
" blob.upload_from_string(content)\n",
"\n",
"\n",
"# Get the task details by using task name.\n",
"def get_task_detail(\n",
" task_details: List[Dict[str, Any]], task_name: str\n",
") -> List[Dict[str, Any]]:\n",
@@ -696,7 +619,6 @@
" return task_detail\n",
"\n",
"\n",
"# Get the model name from pipeline task details.\n",
"def get_model_name(job_id: str) -> str:\n",
" pipeline_task_details = aiplatform.PipelineJob.get(\n",
" job_id\n",
@@ -705,7 +627,6 @@
" return upload_task_details.outputs[\"model\"].artifacts[0].metadata[\"resourceName\"]\n",
"\n",
"\n",
"# Get the evaluation metrics.\n",
"def get_evaluation_metrics(\n",
" task_details: List[Dict[str, Any]],\n",
") -> str:\n",
@@ -760,7 +681,7 @@
"source": [
"### Configure feature transformation\n",
"\n",
"Transformations can be specified using Feature Transform Engine (FTE) specific configurations. Below, you configure full auto transformations (i.e., `auto_transform_config`). FTE automatically configures a set of built-in transformations for each input column based on its data statistics. \n",
"Transformations can be specified using Feature Transform Engine (FTE) specific configurations. Below, we configure full auto transformations (i.e., `auto_transform_config`). FTE automatically configures a set of built-in transformations for each input column based on its data statistics. \n",
"\n",
"For a complete list of supported feature transformation configs and examples, please go [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.15/google_cloud_pipeline_components.experimental.automl.tabular.html#google_cloud_pipeline_components.experimental.automl.tabular.FeatureTransformEngineOp)."
]
@@ -926,7 +847,7 @@
"pipeline_job_root_dir = os.path.join(BUCKET_URI, \"tabnet_custom_job\")\n",
"\n",
"# max_steps and/or max_train_secs must be set. If both are\n",
"# specified, training stop after either condition is met.\n",
"# specified, training will stop after either condition is met.\n",
"# By default, max_train_secs is set to -1.\n",
"\n",
"max_steps = 1000\n",
@@ -989,7 +910,7 @@
" enable_caching=False,\n",
")\n",
"\n",
"pipeline_job.run(service_account=SERVICE_ACCOUNT)"
"pipeline_job.run()"
]
},
{
@@ -1029,9 +950,9 @@
"source": [
"## Customize TabNet HyperparameterTuningJob configuration and create pipeline\n",
"\n",
"To get the best set of hyperparameters for your dataset, you recommend running a HyperparameterTuningJob.\n",
"To get the best set of hyperparameters for your dataset, we recommend running a HyperparameterTuningJob.\n",
"\n",
"Hyperparameters that can be tuned are set in the optional `study_spec_parameters_override` parameter. you provide a helper function called `get_tabnet_study_spec_parameters_override` to get these hyperparameters. You provide `dataset_size_bucket` (one of 'small' (< 1M rows), 'medium' (1M - 100M rows), or 'large' (> 100M rows)), `training_budget_bucket` (one of 'small' (< \\\\$600), 'medium' (\\\\$600 - \\\\$2400), or 'large' (> \\\\$2400)), and `prediction_type` and Vertex AI returns a list of hyperparameters and ranges. `study_spec_parameters_override` can be empty or one or more of these hyperparameters can be specified. For hyperparameters not specified in `study_spec_parameters_override`, you set ranges in the pipeline. For a full list of hyperparameters available for tuning, see [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.23/google_cloud_pipeline_components.experimental.automl.tabular.html#google_cloud_pipeline_components.experimental.automl.tabular.utils.get_tabnet_trainer_pipeline_and_parameters).\n",
"Hyperparameters that can be tuned are set in the optional `study_spec_parameters_override` parameter. We provide a helper function called `get_tabnet_study_spec_parameters_override` to get these hyperparameters. You provide `dataset_size_bucket` (one of 'small' (< 1M rows), 'medium' (1M - 100M rows), or 'large' (> 100M rows)), `training_budget_bucket` (one of 'small' (< \\\\$600), 'medium' (\\\\$600 - \\\\$2400), or 'large' (> \\\\$2400)), and `prediction_type` and Vertex AI returns a list of hyperparameters and ranges. `study_spec_parameters_override` can be empty or one or more of these hyperparameters can be specified. For hyperparameters not specified in `study_spec_parameters_override`, we set ranges in the pipeline. For a full list of hyperparameters available for tuning, see [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.23/google_cloud_pipeline_components.experimental.automl.tabular.html#google_cloud_pipeline_components.experimental.automl.tabular.utils.get_tabnet_trainer_pipeline_and_parameters).\n",
"\n",
"In addition to hyperparameters, HyperparameterTuningJob takes the following values in the example below:\n",
"\n",
@@ -1047,7 +968,7 @@
"\n",
"For a full list of HyperparameterTuningJob parameters, see [here](https://google-cloud-pipeline-components.readthedocs.io/en/google-cloud-pipeline-components-1.0.23/google_cloud_pipeline_components.experimental.automl.tabular.html#google_cloud_pipeline_components.experimental.automl.tabular.utils.get_tabnet_hyperparameter_tuning_job_pipeline_and_parameters).\n",
"\n",
"Multiple trials can be configured. The pipeline returns the best trial based on the metric configured in `study_spec_metrics`. In the example below, you return the trial with the lowest loss value. "
"Multiple trials can be configured. The pipeline returns the best trial based on the metric configured in `study_spec_metrics`. In the example below, we return the trial with the lowest loss value. "
]
},
{
@@ -1079,7 +1000,7 @@
"study_spec_metric_goal = \"MINIMIZE\"\n",
"\n",
"# max_steps and/or max_train_secs must be set. If both are\n",
"# specified, training stop after either condition is met.\n",
"# specified, training will stop after either condition is met.\n",
"# By default, max_train_secs is set to -1 and max_steps is set to\n",
"# an appropriate range given dataset_size and training budget.\n",
"study_spec_parameters_override = (\n",
@@ -1131,7 +1052,7 @@
" enable_caching=False,\n",
")\n",
"\n",
"pipeline_job.run(service_account=SERVICE_ACCOUNT)"
"pipeline_job.run()"
]
},
{
@@ -1156,7 +1077,6 @@
" pipeline_job_id\n",
").gca_resource.job_detail.task_details\n",
"HPT_JOB_MODEL = get_model_name(pipeline_job_id)\n",
"\n",
"print(\"model uri:\", get_model_uri(tabnet_hpt_pipeline_task_details))\n",
"print(\n",
" \"model artifacts:\",\n",
@@ -202,29 +202,9 @@
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install google-cloud-aiplatform {USER_FLAG} -q\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "b24902cde81b"
},
"source": [
"### Restart the kernel\n",
"! pip3 install google-cloud-aiplatform {USER_FLAG} -q\n",
"\n",
"Once you've installed the additional packages, you need to restart the notebook kernel so it can find the packages."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "c61d171395d7"
},
"outputs": [],
"source": [
"import os\n",
"# Automatically restart kernel after installs\n",
"\n",
"if not os.getenv(\"IS_TESTING\"):\n",
" # Automatically restart kernel after installs\n",
@@ -234,13 +214,21 @@
" app.kernel.do_shutdown(True)"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "r_dA3M6UJELw"
},
"source": [
"## Before you begin"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "1Dunp1YrhPYo"
},
"source": [
"## Before you begin\n",
"### Set up your Google Cloud project\n",
"\n",
"**The following steps are required, regardless of your notebook environment.**\n",
@@ -278,17 +266,6 @@
"**If you don't know your project ID**, you may be able to get your project ID using `gcloud`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cde8e0876d62"
},
"outputs": [],
"source": [
"PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
"cell_type": "code",
"execution_count": null,
@@ -297,11 +274,15 @@
},
"outputs": [],
"source": [
"if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n",
" # Get your GCP project id from gcloud\n",
" shell_output = ! gcloud config list --format 'value(core.project)' 2>/dev/null\n",
"PROJECT_ID = \"\"\n",
"\n",
"import os\n",
"\n",
"# Get your Google Cloud project ID from gcloud\n",
"if not os.getenv(\"IS_TESTING\"):\n",
" shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n",
" PROJECT_ID = shell_output[0]\n",
" print(\"Project ID:\", PROJECT_ID)"
" print(\"Project ID: \", PROJECT_ID)"
]
},
{
@@ -321,7 +302,8 @@
},
"outputs": [],
"source": [
"! gcloud config set project $PROJECT_ID"
"if PROJECT_ID == \"\" or PROJECT_ID is None:\n",
" PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}"
]
},
{
@@ -330,9 +312,16 @@
"id": "K-KuU54IaVz5"
},
"source": [
"#### UUID\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"### Timestamp"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "doEJxrvsaWyt"
},
"source": [
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
@@ -343,16 +332,9 @@
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
@@ -361,18 +343,7 @@
"id": "Ee3vBgvdhgTb"
},
"source": [
"#### Region\n",
"\n",
"You can also change the `REGION` variable, which is used for operations\n",
"throughout the rest of this notebook. Below are regions supported for Vertex AI. It is recommended that you choose the region closest to you.\n",
"\n",
"- Americas: `us-central1`\n",
"- Europe: `europe-west4`\n",
"- Asia Pacific: `asia-east1`\n",
"\n",
"You may not use a multi-regional bucket for training with Vertex AI. Not all regions provide support for all Vertex AI services.\n",
"\n",
"Learn more about [Vertex AI regions](https://cloud.google.com/vertex-ai/docs/general/locations)."
"### Set your region"
]
},
{
@@ -384,7 +355,6 @@
"outputs": [],
"source": [
"REGION = \"[your-region]\" # @param {type: \"string\"}\n",
"\n",
"if REGION == \"[your-region]\":\n",
" REGION = \"us-central1\""
]
@@ -395,47 +365,16 @@
"id": "KuNRbXkIijp6"
},
"source": [
"### Authenticate your Google Cloud account\n",
"\n",
"**If you are using Vertex AI Workbench Notebooks**, your environment is already\n",
"authenticated."
"### Login to your Google Cloud account"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "f40aa139740f"
},
"source": [
"**If you are using Colab**, run the cell below and follow the instructions\n",
"when prompted to authenticate your account via oAuth.\n",
"\n",
"**Otherwise**, follow these steps:\n",
"\n",
"1. In the Cloud Console, go to the [**Create service account key**\n",
" page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n",
"\n",
"2. Click **Create service account**.\n",
"\n",
"3. In the **Service account name** field, enter a name, and\n",
" click **Create**.\n",
"\n",
"4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n",
"into the filter box, and select\n",
" **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n",
"\n",
"5. Click *Create*. A JSON file that contains your key downloads to your\n",
"local environment.\n",
"\n",
"6. Enter the path to your service account key as the\n",
"`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell."
]
},
{
"cell_type": "markdown",
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "P9vQxUzfirCV"
},
"outputs": [],
"source": [
"# The Google Cloud Notebook product has specific requirements\n",
"import os\n",
@@ -491,7 +430,7 @@
"BUCKET_URI = \"gs://[your-bucket-name]\" # @param {type:\"string\"}\n",
"\n",
"if BUCKET_URI == \"\" or BUCKET_URI is None or BUCKET_URI == \"gs://[your-bucket-name]\":\n",
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + UUID"
" BUCKET_URI = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP"
]
},
{
@@ -578,25 +517,10 @@
{
"cell_type": "markdown",
"metadata": {
"id": "4eaef8c7be0e"
"id": "0j1NWIQEJI5i"
},
"source": [
"### Enable Artifact Registry API\n",
"First, you must enable the Artifact Registry API service for your project.\n",
"\n",
"Learn more about [Enabling service\n",
" page](https://cloud.google.com/artifact-registry/docs/enable-service)."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "d03035c8fb6f"
},
"outputs": [],
"source": [
"!gcloud services enable artifactregistry.googleapis.com"
"## Create Docker repository"
]
},
{
@@ -605,8 +529,6 @@
"id": "hNmHMIyjBzxx"
},
"source": [
"### Create Docker repository\n",
"\n",
"Create a Docker repository named `DOCKER_REPOSITORY` in your `REGION`.\n",
"This docker repository will be deleted in the clearning up section in the end."
]
@@ -626,7 +548,7 @@
" or DOCKER_REPOSITORY is None\n",
" or DOCKER_REPOSITORY == \"[your-docker-repository-name]\"\n",
"):\n",
" DOCKER_REPOSITORY = \"tb-docker-repo-\" + PROJECT_ID + \"-\" + UUID\n",
" DOCKER_REPOSITORY = \"tb-docker-repo-\" + PROJECT_ID + \"-\" + TIMESTAMP\n",
"\n",
"print(\"Docker repository to create:\", DOCKER_REPOSITORY)"
]
@@ -639,9 +561,18 @@
},
"outputs": [],
"source": [
"! gcloud artifacts repositories create $DOCKER_REPOSITORY --project={PROJECT_ID} \\\n",
"! gcloud artifacts repositories create $DOCKER_REPOSITORY --project={PROJECT_ID} \\\n",
"--repository-format=docker \\\n",
"--location={REGION} --description=\"Repository for TensorBoard Custom Training Job\" "
"--location={REGION} --description=\"Repository for TensorBoard Custom Training Job\""
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "V0pBSC0rDlvq"
},
"source": [
"Verify your Docker repository is created successfully."
]
},
{
@@ -683,7 +614,6 @@
"id": "jUcVG77dKmPn"
},
"source": [
"### Create a training code\n",
"Write your own training code in task.py file. You can use the following code as an example."
]
},
@@ -823,9 +753,7 @@
"id": "DK2E1xz8Q7Q-"
},
"source": [
"Build your container image using `gcloud builds` from your training code and `Dockerfile`. \n",
"\n",
"*Note* that this step may take a few minutes."
"Build your container image using `gcloud builds` from your training code and `Dockerfile`. Note that this step may take a few minutes."
]
},
{
@@ -845,14 +773,21 @@
"! gcloud builds submit --project {PROJECT_ID} --region={REGION} --tag {IMAGE_URI} --timeout=20m"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "hwXxa4Qgnh4Y"
},
"source": [
"## Setup service account and permissions"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "7qXFUiHLoFRw"
},
"source": [
"## Setup service account and permissions\n",
"\n",
"A service account will be used to create custom training job. If you do not want to use your project's Compute Engine service account, set SERVICE_ACCOUNT to another service account ID. You can create a service account by following the [instruction](https://cloud.google.com/iam/docs/creating-managing-service-accounts#creating)."
]
},
@@ -864,7 +799,7 @@
},
"outputs": [],
"source": [
"SERVICE_ACCOUNT = \"[your-service-account]\""
"SERVICE_ACCOUNT = \"[your-service-account]\" # @param {type:\"string\"}"
]
},
{
@@ -875,9 +810,6 @@
},
"outputs": [],
"source": [
"import sys\n",
"\n",
"IS_COLAB = \"google.colab\" in sys.modules\n",
"if (\n",
" SERVICE_ACCOUNT == \"\"\n",
" or SERVICE_ACCOUNT is None\n",
@@ -900,13 +832,37 @@
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "c7798d69970b"
"id": "UlDhuciOt5vo"
},
"outputs": [],
"source": [
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectCreator $BUCKET_URI\n",
"\n",
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.objectViewer $BUCKET_URI"
"# Grant Cloud Storage permission.\n",
"! gcloud projects add-iam-policy-binding {PROJECT_ID} \\\n",
" --member=serviceAccount:{SERVICE_ACCOUNT} \\\n",
" --role=roles/storage.admin"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "lTKVB71soRyr"
},
"outputs": [],
"source": [
"# Grant AI Platform permission.\n",
"! gcloud projects add-iam-policy-binding {PROJECT_ID} \\\n",
" --member=serviceAccount:{SERVICE_ACCOUNT} \\\n",
" --role=roles/aiplatform.user"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "IaQjIPvuKLwW"
},
"source": [
"## Create a custom training job with your container"
]
},
{
@@ -915,7 +871,6 @@
"id": "svUGBOow_Obj"
},
"source": [
"## Create a custom training job with your container\n",
"Create a TensorBoard instnace to be used by the custom training job."
]
},
@@ -934,7 +889,7 @@
" or TENSORBOARD_NAME is None\n",
" or TENSORBOARD_NAME == \"[your-tensorboard-name]\"\n",
"):\n",
" TENSORBOARD_NAME = PROJECT_ID + \"-tb-\" + UUID\n",
" TENSORBOARD_NAME = PROJECT_ID + \"-tb-\" + TIMESTAMP\n",
"\n",
"tensorboard = aiplatform.Tensorboard.create(\n",
" display_name=TENSORBOARD_NAME, project=PROJECT_ID, location=REGION\n",
@@ -960,7 +915,7 @@
},
"outputs": [],
"source": [
"JOB_NAME = \"tensorboard-example-job-{}\".format(UUID)\n",
"JOB_NAME = \"tensorboard-example-job-{}\".format(TIMESTAMP)\n",
"BASE_OUTPUT_DIR = \"{}/{}\".format(BUCKET_URI, JOB_NAME)\n",
"\n",
"job = aiplatform.CustomContainerTrainingJob(\n",
@@ -1009,6 +964,9 @@
},
"outputs": [],
"source": [
"# Delete GCS bucket.\n",
"! gsutil -m rm -r {BUCKET_URI}\n",
"\n",
"# Delete docker repository.\n",
"! gcloud artifacts repositories delete $DOCKER_REPOSITORY --project {PROJECT_ID} --location {REGION} --quiet\n",
"\n",
@@ -1016,12 +974,7 @@
"! gcloud ai tensorboards delete {TENSORBOARD_RESOURCE_NAME}\n",
"\n",
"# Delete custom job.\n",
"job.delete()\n",
"\n",
"# Delete GCS bucket.\n",
"delete_bucket = False\n",
"if delete_bucket or os.getenv(\"IS_TESTING\"):\n",
" ! gsutil -m rm -r $BUCKET_URI"
"job.delete()"
]
}
],
File diff suppressed because it is too large Load Diff
@@ -34,18 +34,18 @@
"<table align=\"left\">\n",
"\n",
" <td>\n",
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/training/hyperparameter_tuning_tensorflow.ipynb\">\n",
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb\">\n",
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/official/training/hyperparameter_tuning_tensorflow.ipynb\">\n",
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/notebook_template.ipynb\">\n",
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
" View on GitHub\n",
" </a>\n",
" </td>\n",
" <td>\n",
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/official/training/hyperparameter_tuning_tensorflow.ipynb\">\n",
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/notebook_template.ipynb\">\n",
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
" Open in Vertex AI Workbench\n",
" </a>\n",
@@ -75,7 +75,7 @@
"id": "dbab58d4ae1a"
},
"source": [
"### Objective\n",
"## Objective\n",
"In this tutorial, you learn how to build ARIMA (Autoregressive integrated moving average) model from BigQuery ML on retail data\n",
"\n",
"This tutorial uses the following Google Cloud ML services:\n",
@@ -97,7 +97,7 @@
"id": "26a00a419045"
},
"source": [
"### Dataset \n",
"## Dataset \n",
"\n",
"This notebook uses the BigQuery public retail data set.\n",
"The data covers 10 US stores and includes item level, department, product categories, and store details. In addition, it has explanatory variables such as price and gross margin. "
@@ -109,7 +109,7 @@
"id": "17e0532066d7"
},
"source": [
"### Costs\n",
"## Costs\n",
"This tutorial uses the following billable components of Google Cloud:\n",
"\n",
"* Vertex AI\n",
@@ -170,7 +170,7 @@
"id": "oH0bZDCmp930"
},
"source": [
"## Install additional packages\n"
"### Install additional packages\n"
]
},
{
@@ -192,9 +192,18 @@
"# Vertex AI Notebook requires dependencies to be installed with '--user'\n",
"USER_FLAG = \"\"\n",
"if IS_WORKBENCH_NOTEBOOK:\n",
" USER_FLAG = \"--user\"\n",
"\n",
"! pip3 install {USER_FLAG} --upgrade pandas-gbq 'google-cloud-bigquery[bqstorage,pandas]' scikit-learn"
" USER_FLAG = \"--user\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "qkHGZ4xYp933"
},
"outputs": [],
"source": [
"! pip3 install {USER_FLAG} --upgrade pandas-gbq 'google-cloud-bigquery[bqstorage,pandas]' sklearn \n"
]
},
{
@@ -252,9 +261,9 @@
"\n",
"1. [Enable the Vertex AI, Cloud Storage, and Compute Engine APIs](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,storage-component.googleapis.com). \n",
"\n",
"1. [Configure your Google Cloud project for Vertex AI Pipelines](https://cloud.google.com/vertex-ai/docs/pipelines/configure-project).\n",
"1. [Configure your Google Cloud project for Vertex Pipelines](https://cloud.google.com/vertex-ai/docs/pipelines/configure-project).\n",
"\n",
"1. If you are running this notebook locally, you need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n",
"\n",
"1. Enter your project ID in the cell below. Then run the cell to make sure the\n",
"Cloud SDK uses the right project for all the commands in this notebook.\n",
@@ -316,9 +325,9 @@
"id": "07fc8daffbf9"
},
"source": [
"#### UUID\n",
"#### Timestamp\n",
"\n",
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a uuid for each instance session, and append it onto the name of resources you create in this tutorial."
"If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial."
]
},
{
@@ -329,16 +338,9 @@
},
"outputs": [],
"source": [
"import random\n",
"import string\n",
"from datetime import datetime\n",
"\n",
"\n",
"# Generate a uuid of a specifed length(default=8)\n",
"def generate_uuid(length: int = 8) -> str:\n",
" return \"\".join(random.choices(string.ascii_lowercase + string.digits, k=length))\n",
"\n",
"\n",
"UUID = generate_uuid()"
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")"
]
},
{
@@ -431,7 +433,7 @@
"id": "b81a2c71fa3a"
},
"source": [
"**Load the required libraries.**"
"Load the required libraries."
]
},
{
@@ -442,6 +444,8 @@
},
"outputs": [],
"source": [
"import datetime\n",
"\n",
"import matplotlib.pyplot as plt\n",
"import pandas as pd\n",
"from google.cloud import bigquery\n",
@@ -463,7 +467,7 @@
"id": "40902aa0f1de"
},
"source": [
"**Set the name for the table**"
"Set the name for the table"
]
},
{
@@ -487,7 +491,7 @@
"id": "36d3a8aec700"
},
"source": [
"**Create a BigQuery datatset**"
"Create a BigQuery datatset"
]
},
{
@@ -498,7 +502,7 @@
},
"outputs": [],
"source": [
"dataset_id = \"demandforecasting\" + \"_\" + UUID"
"dataset_id = \"demandforecasting\" + \"_\" + TIMESTAMP"
]
},
{
@@ -529,7 +533,7 @@
"id": "RxnaBh4sp93_"
},
"source": [
"(**Optional**)If you are using Vertex AI Workbench managed notebooks instance, once the results from BigQuery are displayed in the below cell, click the **Query and load as DataFrame** button and execute the generated code stub to fetch the data into the current notebook as a dataframe.\n",
"(**Optional**)If you are using Vertex AI Workbench managed notebooks instance, once the results from BigQuery are displayed in the above cell, click the **Query and load as DataFrame** button and execute the generated code stub to fetch the data into the current notebook as a dataframe.\n",
"\n",
"*Note: By default the data is loaded into a `df` variable, though this can be changed before executing the cell if required.*"
]
@@ -560,7 +564,15 @@
"id": "7002b223b2b5"
},
"source": [
"## Explore the Data\n",
"## Explore the Data\n"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "6a3ee00b6b9f"
},
"source": [
"View the data that is stored in the public BigQuery dataset."
]
},
@@ -605,7 +617,7 @@
"id": "6d52803e14f2"
},
"source": [
"**Create a view named `important_fields` using only the `transaction_timestamp` and `line_items` fields, where the store ID is 10.**"
"Create a view named `important_fields` using only the `transaction_timestamp` and `line_items` fields, where the store ID is 10."
]
},
{
@@ -647,7 +659,7 @@
"id": "a9ca8a8bbf82"
},
"source": [
"**Look at the data in the `important_fields` view.**"
"Look at the data in the `important_fields` view."
]
},
{
@@ -673,8 +685,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "LTHxGXsGp94B"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -684,7 +705,7 @@
"id": "5f03b43de905"
},
"source": [
"**Convert the `transaction_timestamp` field into a date.**"
"Convert the `transaction_timestamp` field into a date."
]
},
{
@@ -726,7 +747,7 @@
"id": "f0babc4a23fa"
},
"source": [
"**View the data and check the `date` field values.**"
"View the data and check the `date` field values."
]
},
{
@@ -752,8 +773,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "3mtjBiiPp94D"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -763,7 +793,7 @@
"id": "990d04eab2e1"
},
"source": [
"**Load the data into a dataframe.**"
"Load the data into a dataframe."
]
},
{
@@ -783,7 +813,7 @@
"id": "f25cf5322fbc"
},
"source": [
"**Check the data types of your dataframe's fields.**"
"Check the data types of your dataframe's fields."
]
},
{
@@ -845,7 +875,7 @@
"id": "6c677762b34a"
},
"source": [
"**View the data.**"
"View the data."
]
},
{
@@ -871,8 +901,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "cTdONqeAp94F"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -882,7 +921,7 @@
"id": "7acc57476f1c"
},
"source": [
"**Remove the extra columns to keep only `date` and `product_id`.**"
"Remove the extra columns to keep only `date` and `product_id`."
]
},
{
@@ -924,7 +963,7 @@
"id": "4f9d72c483e8"
},
"source": [
"**View the data.**"
"View the data."
]
},
{
@@ -950,8 +989,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "Ao0EHdchp94G"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -961,7 +1009,7 @@
"id": "7c0c4245acb7"
},
"source": [
"**Count the sales of a product for each date.**"
"Count the sales of a product for each date."
]
},
{
@@ -1020,8 +1068,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "7gLx8lkYp94H"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -1031,7 +1088,7 @@
"id": "722fd013c28c"
},
"source": [
"**Create a view for the five products that have sold the most units over the entire date range.**"
"Create a view for the five products that have sold the most units over the entire date range."
]
},
{
@@ -1124,8 +1181,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "aj8MMWkQp94I"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -1135,7 +1201,7 @@
"id": "292855967806"
},
"source": [
"**Load the data into a dataframe and view the data.**"
"Load the data into a dataframe."
]
},
{
@@ -1146,8 +1212,27 @@
},
"outputs": [],
"source": [
"df = query_job.to_dataframe()\n",
"print(df)"
"df = query_job.to_dataframe()"
]
},
{
"cell_type": "markdown",
"metadata": {
"id": "6ebc0bb8ab19"
},
"source": [
"View the data."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "8d11dc64ae2a"
},
"outputs": [],
"source": [
"df"
]
},
{
@@ -1156,7 +1241,7 @@
"id": "5c78641e2881"
},
"source": [
"**Check the data types of your dataframe's fields.**"
"Check the data types of your dataframe's fields."
]
},
{
@@ -1176,7 +1261,7 @@
"id": "ef761dcd0109"
},
"source": [
"**Convert the `date` field's data type to `datetime`.**"
"Convert the `date` field's data type to `datetime`."
]
},
{
@@ -1200,7 +1285,7 @@
"\n",
"To construct a dataframe with `0` values for the `sales_count` field, on dates in which products were not sold, determine the minimum and maximum dates so that you know which dates need `0` values.\n",
"\n",
"**First, get the earliest (minimum) date.**"
"First, get the earliest (minimum) date."
]
},
{
@@ -1226,8 +1311,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "e6FG8Ohcp94K"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -1237,7 +1331,7 @@
"id": "fbb934199011"
},
"source": [
"**Get the latest (maximum) date.**"
"Get the latest (maximum) date."
]
},
{
@@ -1263,8 +1357,17 @@
"\"\"\".format(\n",
" dataset_id=dataset_id\n",
")\n",
"query_job = client.query(query)\n",
"\n",
"query_job = client.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "swjUOJmcp94K"
},
"outputs": [],
"source": [
"query_job.to_dataframe()"
]
},
@@ -1274,7 +1377,7 @@
"id": "247a08d866ef"
},
"source": [
"**Add the full date range of values to a dataframe.**"
"Add the full date range of values to a dataframe."
]
},
{
@@ -1294,7 +1397,7 @@
"id": "b9d3afc9f7cd"
},
"source": [
"**Get a description of the `dates` dataframe.**"
"Get a description of the `dates` dataframe."
]
},
{
@@ -1314,7 +1417,7 @@
"id": "49eb81d44b65"
},
"source": [
"**View the data for one of the products, sorted by date, to show that many dates are not present in the dataset.**"
"View the data for one of the products, sorted by date, to show that many dates are not present in the dataset."
]
},
{
@@ -1365,8 +1468,17 @@
" \"int\"\n",
") # convert sales_count column to integer\n",
"print(\"data after converting for a product with product_id 20552\")\n",
"print(df1)\n",
"\n",
"df1"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "4074c59c4fcc"
},
"outputs": [],
"source": [
"df2 = (\n",
" pd.merge(\n",
" df.loc[df[\"product_id\"] == 13596],\n",
@@ -1384,8 +1496,17 @@
"df2[\"sales_count\"] = df2[\"sales_count\"].astype(\n",
" \"int\"\n",
") # convert sales_count column to integer\n",
"print(df2)\n",
"\n",
"df2"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "589ffdbf42f1"
},
"outputs": [],
"source": [
"df3 = (\n",
" pd.merge(\n",
" df.loc[df[\"product_id\"] == 23641],\n",
@@ -1403,8 +1524,17 @@
"df3[\"sales_count\"] = df3[\"sales_count\"].astype(\n",
" \"int\"\n",
") # convert sales_count column to integer\n",
"print(df3)\n",
"\n",
"df3"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "df19f886f7b7"
},
"outputs": [],
"source": [
"df4 = (\n",
" pd.merge(\n",
" df.loc[df[\"product_id\"] == 28305],\n",
@@ -1422,8 +1552,17 @@
"df4[\"sales_count\"] = df4[\"sales_count\"].astype(\n",
" \"int\"\n",
") # convert sales_count column to integer\n",
"print(df4)\n",
"\n",
"df4"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "2bde5882369e"
},
"outputs": [],
"source": [
"df5 = (\n",
" pd.merge(\n",
" df.loc[df[\"product_id\"] == 20547],\n",
@@ -1441,7 +1580,7 @@
"df5[\"sales_count\"] = df5[\"sales_count\"].astype(\n",
" \"int\"\n",
") # convert sales_count column to integer\n",
"print(df5)"
"df5"
]
},
{
@@ -1450,7 +1589,7 @@
"id": "6a2d033b733e"
},
"source": [
"**Merge all five dataframes into one new dataframe.**"
"Merge all five dataframes into one new dataframe"
]
},
{
@@ -1472,7 +1611,7 @@
"id": "4a75fa1dc8dc"
},
"source": [
"**Reset the index of the dataframe.**"
"Reset the index of the dataframe."
]
},
{
@@ -1483,8 +1622,18 @@
},
"outputs": [],
"source": [
"new_df.reset_index(inplace=True, drop=True)\n",
"print(new_df)"
"new_df.reset_index(inplace=True, drop=True)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {
"id": "5312dfe1be15"
},
"outputs": [],
"source": [
"new_df"
]
},
{
@@ -1493,7 +1642,7 @@
"id": "97e1289b2106"
},
"source": [
"**View the five product IDs.**"
"View the five product IDs."
]
},
{
@@ -1539,7 +1688,7 @@
"id": "aa06b0e893cb"
},
"source": [
"**Plot `sales_count` over time, for each product.**"
"Plot `sales_count` over time, for each product."
]
},
{
@@ -1623,7 +1772,7 @@
"id": "755a811fbad5"
},
"source": [
"**List the data types for the `new_df` dataframe.**"
"List the data types for the `new_df` dataframe."
]
},
{
@@ -1643,7 +1792,7 @@
"id": "ff06b56ffa7a"
},
"source": [
"**Create a new BigQuery table out of the `new_df` dataframe.**"
"Create a new BigQuery table out of the `new_df` dataframe."
]
},
{
@@ -1683,7 +1832,7 @@
"id": "c2e0e9aa67cd"
},
"source": [
"**Create a training dataset by setting a date range that limits the data being used.**"
"Create a training dataset by setting a date range that limits the data being used."
]
},
{
@@ -1724,7 +1873,7 @@
"id": "2f7d9d2d4229"
},
"source": [
"**Select the original data for plotting.**"
"Select the original data for plotting."
]
},
{
@@ -1750,7 +1899,7 @@
"source": [
"## Modeling with BigQuery and the ARIMA model\n",
"\n",
"**Create an ARIMA model using the training data.**"
"Create an ARIMA model using the training data."
]
},
{
@@ -1785,7 +1934,7 @@
"id": "c45e18a773ad"
},
"source": [
"**Train the ARIMA model.**"
"Train the ARIMA model."
]
},
{
@@ -1859,7 +2008,7 @@
"id": "801528e3e2c7"
},
"source": [
"**Load the data into a dataframe named `dfforecast`.**"
"Load the data into a dataframe named `dfforecast`."
]
},
{
@@ -1894,7 +2043,7 @@
"id": "1e0549381849"
},
"source": [
"**View the first few rows.**"
"View the first few rows."
]
},
{
@@ -1925,7 +2074,7 @@
"id": "6fa36b762a33"
},
"source": [
"**Clean the historical and forecasted values for plotting.**"
"Clean the historical and forecasted values for plotting."
]
},
{
@@ -1953,7 +2102,7 @@
"id": "d40a95ad0616"
},
"source": [
"**Plot the historical and forecast data.**\n"
"Plot the historical and forecast data.\n"
]
},
{